diff --git a/.github/ISSUE_TEMPLATE/english.md b/.github/ISSUE_TEMPLATE/english.md deleted file mode 100644 index 56cebfba0f8..00000000000 --- a/.github/ISSUE_TEMPLATE/english.md +++ /dev/null @@ -1,18 +0,0 @@ ---- -name: English -about: Report issue in English -title: '' -labels: '' -assignees: '' - ---- - -## Environment - -FastDeploy version: e.g 0.8.0 or the latest code in develop branch -OS Platform: e.g. Linux x64 / Windows x64 / Mac OSX 12.1(arm or intel) -Hardware: e.g. Nvidia GPU 3080Ti CUDA 11.2 CUDNN 8.3 -Program Language: e.g. Python 3.8 - -## Problem description -Please attach the log file if there's problem happend. diff --git a/.github/ISSUE_TEMPLATE/other.md b/.github/ISSUE_TEMPLATE/other.md deleted file mode 100644 index 8b7f46622e8..00000000000 --- a/.github/ISSUE_TEMPLATE/other.md +++ /dev/null @@ -1,10 +0,0 @@ ---- -name: Other -about: Other issues, e.g feature/model request -title: '' -labels: '' -assignees: '' - ---- - - diff --git "a/.github/ISSUE_TEMPLATE/\346\212\245\345\221\212issue.md" "b/.github/ISSUE_TEMPLATE/\346\212\245\345\221\212issue.md" deleted file mode 100755 index cafba1dbc43..00000000000 --- "a/.github/ISSUE_TEMPLATE/\346\212\245\345\221\212issue.md" +++ /dev/null @@ -1,35 +0,0 @@ ---- -name: 报告issue -about: 反馈使用中遇到的问题 -title: '' -labels: '' -assignees: '' - ---- - -********************************************* -温馨提示:根据社区不完全统计,按照模板提问,可以加快回复和解决问题的速度 -********************************************* - -## 环境 - -- 【FastDeploy版本】: 说明具体的版本,如fastdeploy-linux-gpu-0.8.0 -- 【编译命令】如果您是自行编译的FastDeploy,请说明您的编译方式(参数命令) -- 【系统平台】: Linux x64(Ubuntu 18.04) / Windows x64(Windows10) / Mac OSX arm(12.0) / Mac OSX intel(12.0) -- 【硬件】: 说明具体硬件型号,如 Nvidia GPU 3080TI, CUDA 11.2 CUDNN 8.3 -- 【编译语言】: C++ / Python(3.7或3.8等) - -## 问题日志及出现问题的操作流程 -- 附上详细的问题日志有助于快速定位分析 -- 【模型跑不通】 -- - 先执行`examples`下的部署示例,包括使用examples提供的模型,确认是否可以正确执行 -- - 如若`examples`下的代码可以运行,但自己的模型,或自己的代码不能运行 -- - - 提供复现问题的 代码+模型+错误log,供工程师快速定位问题 -- 【模型精度问题】 -- - 先执行`examples`下的部署示例,包括使用examples提供的模型,确认是否可以正确执行 -- - 如若`examples`下的代码可以运行,但自己的模型,或自己的代码不能运行 -- - - 提供复现问题的 代码+模型+错误log,供工程师快速定位问题 -- 【性能问题】描述清楚对比的方式 -- - 注意性能测试,循环跑N次,取后80%的用时平均(模型启动时,刚开始受限于资源分配,速度会较慢) -- - FastDeploy的Predict包含模型本身之外的数据前后处理用时 -- - - 提供复现问题的 代码+模型+错误log,供工程师快速定位问题 diff --git a/.github/PULL_REQUEST_TEMPLATE.md b/.github/PULL_REQUEST_TEMPLATE.md deleted file mode 100644 index 1b4727fd330..00000000000 --- a/.github/PULL_REQUEST_TEMPLATE.md +++ /dev/null @@ -1,10 +0,0 @@ - - - -### PR types(PR类型) - - -### Description - - - diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index d1569f65a1b..f5164e31436 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -1,31 +1,89 @@ -name: Build -on: [pull_request] +name: Build and Package + +on: + push: + pull_request: + branches: [ dev ] + workflow_dispatch: + +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true jobs: - macOS-latest-py: + # ========================================================================= + # 1. macOS ARM64 (Apple Silicon: CoreML & CPU) + # ========================================================================= + build-macos-arm64: + name: macOS (arm64) runs-on: macos-latest + steps: + - uses: actions/checkout@v7 + - uses: lukka/get-cmake@latest + - name: Install System Dependencies + run: brew install opencv@4 eigen onnxruntime + - name: Build & Package + run: python3 scripts/build.py --target macos-arm64 + - uses: actions/upload-artifact@v7 + with: + name: fastdeploy_ppocr-macos-arm64 + path: fastdeploy_ppocr-macos-arm64.tar.gz + # ========================================================================= + # 2. Linux x86_64 (CPU) + # ========================================================================= + build-linux-x64: + name: Linux (x64) + runs-on: ubuntu-latest steps: - - name: Clone - uses: actions/checkout@v1 + - uses: actions/checkout@v7 + - uses: lukka/get-cmake@latest + - name: Install System Dependencies + run: sudo apt-get update && sudo apt-get install -y libopencv-dev libeigen3-dev wget tar + - name: Build & Package + run: python3 scripts/build.py --target linux-x64 + - uses: actions/upload-artifact@v7 + with: + name: fastdeploy_ppocr-linux-x64 + path: fastdeploy_ppocr-linux-x64.tar.gz + + # ========================================================================= + # 3. Windows x64 (DirectML & CPU) + # ========================================================================= + build-windows-x64: + name: Windows (x64) + runs-on: windows-latest + steps: + - name: Enable Git LongPaths + run: git config --system core.longpaths true + - uses: actions/checkout@v7 + - uses: TheMrMilchmann/setup-msvc-dev@v4 + with: + arch: x64 + - uses: lukka/get-cmake@latest + - name: Build & Package + run: python scripts/build.py --target windows-x64 + - uses: actions/upload-artifact@v7 + with: + name: fastdeploy_ppocr-windows-x64 + path: fastdeploy_ppocr-windows-x64.zip - - name: Get CMake - uses: lukka/get-cmake@latest - - name: Get Python - uses: actions/setup-python@v4 + # ========================================================================= + # 5. Android ARM64-v8a (Google NDK Cross Compilation) + # ========================================================================= + build-android-arm64: + name: Android (arm64) + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v7 + - uses: lukka/get-cmake@latest + - name: Install System Dependencies + run: sudo apt-get update && sudo apt-get install -y libeigen3-dev wget unzip + - name: Build & Package + run: python3 scripts/build.py --target android-arm64 + - uses: actions/upload-artifact@v7 with: - python-version: '3.10' - - - name: Build FastDeploy - working-directory: ./python - run: | - export ENABLE_ORT_BACKEND=ON - export ENABLE_PADDLE_BACKEND=OFF - export ENABLE_OPENVINO_BACKEND=OFF - export ENABLE_VISION=ON - export ENABLE_TEXT=ON - python -m pip install wheel - python setup.py build - python setup.py bdist_wheel - ls -l + name: fastdeploy_ppocr-android-arm64 + path: fastdeploy_ppocr-android-arm64.tar.gz + diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index 6da98641596..563c71027f2 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -10,21 +10,12 @@ repos: - id: check-symlinks - id: check-added-large-files -- repo: local - hooks: - - id: copyright_checker - name: copyright_checker - entry: python ./.copyright.hook - language: system - files: \.(c|cc|cxx|cpp|cu|h|hpp|hxx|proto|py)$ - exclude: (?!.*third_party)^.*$ - - repo: local hooks: - id: clang-format-with-version-check name: clang-format description: Format files with ClangFormat. - entry: bash .clang_format.hook -i + entry: bash scripts/hooks/.clang_format.hook -i language: system files: \.(c|cc|cxx|cpp|cu|hxx|proto)$ @@ -33,6 +24,6 @@ repos: - id: cpplint-cpp-source name: cpplint description: Check C++ code style using cpplint.py. - entry: bash .cpplint_pre_commit.hook + entry: bash scripts/hooks/.cpplint_pre_commit.hook language: system files: \.(c|cc|cxx|cpp|cu|h|hpp|hxx)$ diff --git a/CMakeLists.txt b/CMakeLists.txt index 293654a40e4..f662b8c22e5 100755 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -8,7 +8,7 @@ find_package(onnxruntime REQUIRED) find_package(Eigen3 REQUIRED) message("OpenCV_LIBS: ${OpenCV_LIBS} ${OPENCV_core_FOUND} ${OPENCV_WORLD_FOUND}") -option(WITH_CUDA "Whether WITH_CUDA=ON, will enable onnxruntime-gpu/paddle-infernce-gpu/poros-gpu" OFF) +option(WITH_CUDA "Whether WITH_CUDA=ON, will enable onnxruntime-gpu/paddle-inference-gpu" OFF) option(PRINT_INFO "Print more debug info while running" OFF) if(WITH_CUDA) @@ -132,7 +132,7 @@ install(TARGETS fastdeploy_ppocr EXPORT fastdeploy_ppocrConfig) install(EXPORT fastdeploy_ppocrConfig DESTINATION share/fastdeploy_ppocr) install( DIRECTORY ${PROJECT_SOURCE_DIR}/fastdeploy - DESTINATION ${CMAKE_INSTALL_PREFIX}/include + DESTINATION include FILES_MATCHING PATTERN "*.h" ) \ No newline at end of file diff --git a/README.md b/README.md deleted file mode 120000 index b02d435c6f9..00000000000 --- a/README.md +++ /dev/null @@ -1,16 +0,0 @@ -# FastDeploy - -This is a stripped-down version of [PaddlePaddle/FastDeploy](https://github.com/PaddlePaddle/FastDeploy) and [MaaAssistantArknights/FastDeploy](https://github.com/MaaAssistantArknights/FastDeploy). - -## Key modifications - -### For PaddlePaddle/FastDeploy - -* ~~Removed unused components~~ -* Use system library (CMake `find_package`) only -* Library name changed to `fastdeploy_ppocr` - -### For MaaAssistantArknights/FastDeploy - -* Connect to historical commits for later updates -* Fixed reading of files instead of passing them through memory diff --git a/README.md b/README.md new file mode 100644 index 00000000000..ab22e3a63df --- /dev/null +++ b/README.md @@ -0,0 +1,12 @@ +# FastDeploy + +This is a stripped-down version of [PaddlePaddle/FastDeploy@v1](https://github.com/PaddlePaddle/FastDeploy/tree/release/1.1.0) and [MaaAssistantArknights/FastDeploy](https://github.com/MaaAssistantArknights/FastDeploy). + +## Key modifications + +* ~~Removed unused components and dead code, focusing on `fastdeploy_ppocr`~~ +* Use system libraries (CMake `find_package`) only +* Support in-memory model and dictionary loading +* Reorganized project layout (`bindings/`, `services/`) +* Add support for PP-OCR v5 and v6 +* Add multi-platform build script (`scripts/build.py`) \ No newline at end of file diff --git a/VERSION_NUMBER b/VERSION_NUMBER deleted file mode 100644 index 77d6f4ca237..00000000000 --- a/VERSION_NUMBER +++ /dev/null @@ -1 +0,0 @@ -0.0.0 diff --git a/benchmark/python/benchmark_ppocr.py b/benchmark/python/benchmark_ppocr.py index 1fd59a088dd..47be9d2d2b3 100755 --- a/benchmark/python/benchmark_ppocr.py +++ b/benchmark/python/benchmark_ppocr.py @@ -341,6 +341,54 @@ def cpu_stat_func(self, q, pid, interval=0.0): runtime_option=rec_option) model = fd.vision.ocr.PPOCRv4( det_model=det_model, cls_model=cls_model, rec_model=rec_model) + elif "OCRv5" in args.model_dir: + det_option = option + if args.backend in ["trt", "paddle_trt"]: + det_option.trt_option.set_shape( + "x", [1, 3, 64, 64], [1, 3, 640, 640], [1, 3, 960, 960]) + det_model = fd.vision.ocr.DBDetector( + det_model_file, det_params_file, runtime_option=det_option) + cls_option = option + if args.backend in ["trt", "paddle_trt"]: + cls_option.trt_option.set_shape( + "x", [1, 3, 48, 10], [10, 3, 48, 320], [64, 3, 48, 1024]) + cls_model = fd.vision.ocr.Classifier( + cls_model_file, cls_params_file, runtime_option=cls_option) + rec_option = option + if args.backend in ["trt", "paddle_trt"]: + rec_option.trt_option.set_shape( + "x", [1, 3, 48, 10], [10, 3, 48, 320], [64, 3, 48, 2304]) + rec_model = fd.vision.ocr.Recognizer( + rec_model_file, + rec_params_file, + rec_label_file, + runtime_option=rec_option) + model = fd.vision.ocr.PPOCRv5( + det_model=det_model, cls_model=cls_model, rec_model=rec_model) + elif "OCRv6" in args.model_dir: + det_option = option + if args.backend in ["trt", "paddle_trt"]: + det_option.trt_option.set_shape( + "x", [1, 3, 64, 64], [1, 3, 640, 640], [1, 3, 960, 960]) + det_model = fd.vision.ocr.DBDetector( + det_model_file, det_params_file, runtime_option=det_option) + cls_option = option + if args.backend in ["trt", "paddle_trt"]: + cls_option.trt_option.set_shape( + "x", [1, 3, 48, 10], [10, 3, 48, 320], [64, 3, 48, 1024]) + cls_model = fd.vision.ocr.Classifier( + cls_model_file, cls_params_file, runtime_option=cls_option) + rec_option = option + if args.backend in ["trt", "paddle_trt"]: + rec_option.trt_option.set_shape( + "x", [1, 3, 48, 10], [10, 3, 48, 320], [64, 3, 48, 2304]) + rec_model = fd.vision.ocr.Recognizer( + rec_model_file, + rec_params_file, + rec_label_file, + runtime_option=rec_option) + model = fd.vision.ocr.PPOCRv6( + det_model=det_model, cls_model=cls_model, rec_model=rec_model) else: raise Exception("model {} not support now in ppocr series".format( args.model_dir)) diff --git a/c_api/CMakeLists.txt b/bindings/c_api/CMakeLists.txt similarity index 66% rename from c_api/CMakeLists.txt rename to bindings/c_api/CMakeLists.txt index b75bde8fdd4..2f4d56497d7 100644 --- a/c_api/CMakeLists.txt +++ b/bindings/c_api/CMakeLists.txt @@ -19,11 +19,11 @@ if(NOT WITH_CAPI) return() endif() -configure_file(${PROJECT_SOURCE_DIR}/${CSRCS_DIR_NAME}/c_api/fastdeploy_capi/core/config.h.in ${PROJECT_SOURCE_DIR}/${CSRCS_DIR_NAME}/c_api/fastdeploy_capi/core/config.h) -file(GLOB_RECURSE DEPLOY_CAPI_SRCS ${PROJECT_SOURCE_DIR}/${CSRCS_DIR_NAME}/c_api/fastdeploy_capi/*.cc) +configure_file(${CMAKE_CURRENT_SOURCE_DIR}/fastdeploy_capi/core/config.h.in ${CMAKE_CURRENT_SOURCE_DIR}/fastdeploy_capi/core/config.h) +file(GLOB_RECURSE DEPLOY_CAPI_SRCS ${CMAKE_CURRENT_SOURCE_DIR}/fastdeploy_capi/*.cc) if(NOT ENABLE_VISION) - file(GLOB_RECURSE DEPLOY_VISION_CAPI_SRCS ${PROJECT_SOURCE_DIR}/${CSRCS_DIR_NAME}/c_api/fastdeploy_capi/vision/*.cc) + file(GLOB_RECURSE DEPLOY_VISION_CAPI_SRCS ${CMAKE_CURRENT_SOURCE_DIR}/fastdeploy_capi/vision/*.cc) list(REMOVE_ITEM DEPLOY_CAPI_SRCS ${DEPLOY_VISION_CAPI_SRCS}) endif() list(APPEND ALL_DEPLOY_SRCS ${DEPLOY_CAPI_SRCS}) -include_directories(${PROJECT_SOURCE_DIR}/${CSRCS_DIR_NAME}/c_api) +include_directories(${CMAKE_CURRENT_SOURCE_DIR}) diff --git a/c_api/README.md b/bindings/c_api/README.md similarity index 100% rename from c_api/README.md rename to bindings/c_api/README.md diff --git a/c_api/README_CN.md b/bindings/c_api/README_CN.md similarity index 100% rename from c_api/README_CN.md rename to bindings/c_api/README_CN.md diff --git a/c_api/fastdeploy_capi/core/config.h b/bindings/c_api/fastdeploy_capi/core/config.h similarity index 100% rename from c_api/fastdeploy_capi/core/config.h rename to bindings/c_api/fastdeploy_capi/core/config.h diff --git a/c_api/fastdeploy_capi/core/config.h.in b/bindings/c_api/fastdeploy_capi/core/config.h.in similarity index 100% rename from c_api/fastdeploy_capi/core/config.h.in rename to bindings/c_api/fastdeploy_capi/core/config.h.in diff --git a/c_api/fastdeploy_capi/core/fd_common.h b/bindings/c_api/fastdeploy_capi/core/fd_common.h similarity index 100% rename from c_api/fastdeploy_capi/core/fd_common.h rename to bindings/c_api/fastdeploy_capi/core/fd_common.h diff --git a/c_api/fastdeploy_capi/core/fd_type.cc b/bindings/c_api/fastdeploy_capi/core/fd_type.cc similarity index 100% rename from c_api/fastdeploy_capi/core/fd_type.cc rename to bindings/c_api/fastdeploy_capi/core/fd_type.cc diff --git a/c_api/fastdeploy_capi/core/fd_type.h b/bindings/c_api/fastdeploy_capi/core/fd_type.h similarity index 100% rename from c_api/fastdeploy_capi/core/fd_type.h rename to bindings/c_api/fastdeploy_capi/core/fd_type.h diff --git a/c_api/fastdeploy_capi/internal/types_internal.cc b/bindings/c_api/fastdeploy_capi/internal/types_internal.cc similarity index 100% rename from c_api/fastdeploy_capi/internal/types_internal.cc rename to bindings/c_api/fastdeploy_capi/internal/types_internal.cc diff --git a/c_api/fastdeploy_capi/internal/types_internal.h b/bindings/c_api/fastdeploy_capi/internal/types_internal.h similarity index 100% rename from c_api/fastdeploy_capi/internal/types_internal.h rename to bindings/c_api/fastdeploy_capi/internal/types_internal.h diff --git a/c_api/fastdeploy_capi/runtime/enum_variables.h b/bindings/c_api/fastdeploy_capi/runtime/enum_variables.h similarity index 100% rename from c_api/fastdeploy_capi/runtime/enum_variables.h rename to bindings/c_api/fastdeploy_capi/runtime/enum_variables.h diff --git a/c_api/fastdeploy_capi/runtime/runtime_option.cc b/bindings/c_api/fastdeploy_capi/runtime/runtime_option.cc similarity index 100% rename from c_api/fastdeploy_capi/runtime/runtime_option.cc rename to bindings/c_api/fastdeploy_capi/runtime/runtime_option.cc diff --git a/c_api/fastdeploy_capi/runtime/runtime_option.h b/bindings/c_api/fastdeploy_capi/runtime/runtime_option.h similarity index 100% rename from c_api/fastdeploy_capi/runtime/runtime_option.h rename to bindings/c_api/fastdeploy_capi/runtime/runtime_option.h diff --git a/c_api/fastdeploy_capi/vision.h b/bindings/c_api/fastdeploy_capi/vision.h similarity index 100% rename from c_api/fastdeploy_capi/vision.h rename to bindings/c_api/fastdeploy_capi/vision.h diff --git a/c_api/fastdeploy_capi/vision/classification/ppcls/model.cc b/bindings/c_api/fastdeploy_capi/vision/classification/ppcls/model.cc similarity index 100% rename from c_api/fastdeploy_capi/vision/classification/ppcls/model.cc rename to bindings/c_api/fastdeploy_capi/vision/classification/ppcls/model.cc diff --git a/c_api/fastdeploy_capi/vision/classification/ppcls/model.h b/bindings/c_api/fastdeploy_capi/vision/classification/ppcls/model.h similarity index 100% rename from c_api/fastdeploy_capi/vision/classification/ppcls/model.h rename to bindings/c_api/fastdeploy_capi/vision/classification/ppcls/model.h diff --git a/c_api/fastdeploy_capi/vision/detection/contrib/yolo/base_define.h b/bindings/c_api/fastdeploy_capi/vision/detection/contrib/yolo/base_define.h similarity index 100% rename from c_api/fastdeploy_capi/vision/detection/contrib/yolo/base_define.h rename to bindings/c_api/fastdeploy_capi/vision/detection/contrib/yolo/base_define.h diff --git a/c_api/fastdeploy_capi/vision/detection/contrib/yolo/model.cc b/bindings/c_api/fastdeploy_capi/vision/detection/contrib/yolo/model.cc similarity index 100% rename from c_api/fastdeploy_capi/vision/detection/contrib/yolo/model.cc rename to bindings/c_api/fastdeploy_capi/vision/detection/contrib/yolo/model.cc diff --git a/c_api/fastdeploy_capi/vision/detection/contrib/yolo/model.h b/bindings/c_api/fastdeploy_capi/vision/detection/contrib/yolo/model.h similarity index 100% rename from c_api/fastdeploy_capi/vision/detection/contrib/yolo/model.h rename to bindings/c_api/fastdeploy_capi/vision/detection/contrib/yolo/model.h diff --git a/c_api/fastdeploy_capi/vision/detection/ppdet/base_define.h b/bindings/c_api/fastdeploy_capi/vision/detection/ppdet/base_define.h similarity index 100% rename from c_api/fastdeploy_capi/vision/detection/ppdet/base_define.h rename to bindings/c_api/fastdeploy_capi/vision/detection/ppdet/base_define.h diff --git a/c_api/fastdeploy_capi/vision/detection/ppdet/model.cc b/bindings/c_api/fastdeploy_capi/vision/detection/ppdet/model.cc similarity index 100% rename from c_api/fastdeploy_capi/vision/detection/ppdet/model.cc rename to bindings/c_api/fastdeploy_capi/vision/detection/ppdet/model.cc diff --git a/c_api/fastdeploy_capi/vision/detection/ppdet/model.h b/bindings/c_api/fastdeploy_capi/vision/detection/ppdet/model.h similarity index 100% rename from c_api/fastdeploy_capi/vision/detection/ppdet/model.h rename to bindings/c_api/fastdeploy_capi/vision/detection/ppdet/model.h diff --git a/c_api/fastdeploy_capi/vision/ocr/ppocr/base_define.h b/bindings/c_api/fastdeploy_capi/vision/ocr/ppocr/base_define.h similarity index 100% rename from c_api/fastdeploy_capi/vision/ocr/ppocr/base_define.h rename to bindings/c_api/fastdeploy_capi/vision/ocr/ppocr/base_define.h diff --git a/c_api/fastdeploy_capi/vision/ocr/ppocr/model.cc b/bindings/c_api/fastdeploy_capi/vision/ocr/ppocr/model.cc similarity index 100% rename from c_api/fastdeploy_capi/vision/ocr/ppocr/model.cc rename to bindings/c_api/fastdeploy_capi/vision/ocr/ppocr/model.cc diff --git a/c_api/fastdeploy_capi/vision/ocr/ppocr/model.h b/bindings/c_api/fastdeploy_capi/vision/ocr/ppocr/model.h similarity index 100% rename from c_api/fastdeploy_capi/vision/ocr/ppocr/model.h rename to bindings/c_api/fastdeploy_capi/vision/ocr/ppocr/model.h diff --git a/c_api/fastdeploy_capi/vision/result.cc b/bindings/c_api/fastdeploy_capi/vision/result.cc similarity index 100% rename from c_api/fastdeploy_capi/vision/result.cc rename to bindings/c_api/fastdeploy_capi/vision/result.cc diff --git a/c_api/fastdeploy_capi/vision/result.h b/bindings/c_api/fastdeploy_capi/vision/result.h similarity index 100% rename from c_api/fastdeploy_capi/vision/result.h rename to bindings/c_api/fastdeploy_capi/vision/result.h diff --git a/c_api/fastdeploy_capi/vision/segmentation/ppseg/model.cc b/bindings/c_api/fastdeploy_capi/vision/segmentation/ppseg/model.cc similarity index 100% rename from c_api/fastdeploy_capi/vision/segmentation/ppseg/model.cc rename to bindings/c_api/fastdeploy_capi/vision/segmentation/ppseg/model.cc diff --git a/c_api/fastdeploy_capi/vision/segmentation/ppseg/model.h b/bindings/c_api/fastdeploy_capi/vision/segmentation/ppseg/model.h similarity index 100% rename from c_api/fastdeploy_capi/vision/segmentation/ppseg/model.h rename to bindings/c_api/fastdeploy_capi/vision/segmentation/ppseg/model.h diff --git a/c_api/fastdeploy_capi/vision/types_internal.cc b/bindings/c_api/fastdeploy_capi/vision/types_internal.cc similarity index 100% rename from c_api/fastdeploy_capi/vision/types_internal.cc rename to bindings/c_api/fastdeploy_capi/vision/types_internal.cc diff --git a/c_api/fastdeploy_capi/vision/types_internal.h b/bindings/c_api/fastdeploy_capi/vision/types_internal.h similarity index 97% rename from c_api/fastdeploy_capi/vision/types_internal.h rename to bindings/c_api/fastdeploy_capi/vision/types_internal.h index ec8aaf76605..f55b3c6e476 100755 --- a/c_api/fastdeploy_capi/vision/types_internal.h +++ b/bindings/c_api/fastdeploy_capi/vision/types_internal.h @@ -191,6 +191,12 @@ DEFINE_PIPELINE_MODEL_WRAPPER_STRUCT(PPOCRv3, ppocrv3_model); // PPOCRv4 DEFINE_PIPELINE_MODEL_WRAPPER_STRUCT(PPOCRv4, ppocrv4_model); +// PPOCRv5 +DEFINE_PIPELINE_MODEL_WRAPPER_STRUCT(PPOCRv5, ppocrv5_model); + +// PPOCRv6 +DEFINE_PIPELINE_MODEL_WRAPPER_STRUCT(PPOCRv6, ppocrv6_model); + // PPStructureV2Table DEFINE_PIPELINE_MODEL_WRAPPER_STRUCT(PPStructureV2Table, ppstructurev2table_model); @@ -407,6 +413,12 @@ DECLARE_PIPELINE_MODEL_FUNC_FOR_GET_PTR_FROM_WRAPPER(PPOCRv3, fd_ppocrv3_wrapper // PPOCRv4 DECLARE_PIPELINE_MODEL_FUNC_FOR_GET_PTR_FROM_WRAPPER(PPOCRv4, fd_ppocrv4_wrapper); +// PPOCRv5 +DECLARE_PIPELINE_MODEL_FUNC_FOR_GET_PTR_FROM_WRAPPER(PPOCRv5, fd_ppocrv5_wrapper); + +// PPOCRv6 +DECLARE_PIPELINE_MODEL_FUNC_FOR_GET_PTR_FROM_WRAPPER(PPOCRv6, fd_ppocrv6_wrapper); + // PPStructureV2Table DECLARE_PIPELINE_MODEL_FUNC_FOR_GET_PTR_FROM_WRAPPER(PPStructureV2Table, fd_ppstructurev2_table_wrapper); diff --git a/c_api/fastdeploy_capi/vision/visualize.cc b/bindings/c_api/fastdeploy_capi/vision/visualize.cc similarity index 100% rename from c_api/fastdeploy_capi/vision/visualize.cc rename to bindings/c_api/fastdeploy_capi/vision/visualize.cc diff --git a/c_api/fastdeploy_capi/vision/visualize.h b/bindings/c_api/fastdeploy_capi/vision/visualize.h similarity index 100% rename from c_api/fastdeploy_capi/vision/visualize.h rename to bindings/c_api/fastdeploy_capi/vision/visualize.h diff --git a/csharp/CMakeLists.txt b/bindings/csharp/CMakeLists.txt similarity index 100% rename from csharp/CMakeLists.txt rename to bindings/csharp/CMakeLists.txt diff --git a/csharp/README.md b/bindings/csharp/README.md similarity index 100% rename from csharp/README.md rename to bindings/csharp/README.md diff --git a/csharp/README_CN.md b/bindings/csharp/README_CN.md similarity index 100% rename from csharp/README_CN.md rename to bindings/csharp/README_CN.md diff --git a/csharp/fastdeploy/enum_varaibles.cs b/bindings/csharp/fastdeploy/enum_varaibles.cs similarity index 100% rename from csharp/fastdeploy/enum_varaibles.cs rename to bindings/csharp/fastdeploy/enum_varaibles.cs diff --git a/csharp/fastdeploy/runtime_option.cs b/bindings/csharp/fastdeploy/runtime_option.cs similarity index 100% rename from csharp/fastdeploy/runtime_option.cs rename to bindings/csharp/fastdeploy/runtime_option.cs diff --git a/csharp/fastdeploy/types_internal_c.cs b/bindings/csharp/fastdeploy/types_internal_c.cs similarity index 100% rename from csharp/fastdeploy/types_internal_c.cs rename to bindings/csharp/fastdeploy/types_internal_c.cs diff --git a/csharp/fastdeploy/vision/classification/ppcls/model.cs b/bindings/csharp/fastdeploy/vision/classification/ppcls/model.cs similarity index 100% rename from csharp/fastdeploy/vision/classification/ppcls/model.cs rename to bindings/csharp/fastdeploy/vision/classification/ppcls/model.cs diff --git a/csharp/fastdeploy/vision/detection/contrib/model.cs b/bindings/csharp/fastdeploy/vision/detection/contrib/model.cs similarity index 100% rename from csharp/fastdeploy/vision/detection/contrib/model.cs rename to bindings/csharp/fastdeploy/vision/detection/contrib/model.cs diff --git a/csharp/fastdeploy/vision/detection/ppdet/model.cs b/bindings/csharp/fastdeploy/vision/detection/ppdet/model.cs similarity index 100% rename from csharp/fastdeploy/vision/detection/ppdet/model.cs rename to bindings/csharp/fastdeploy/vision/detection/ppdet/model.cs diff --git a/csharp/fastdeploy/vision/ocr/model.cs b/bindings/csharp/fastdeploy/vision/ocr/model.cs similarity index 100% rename from csharp/fastdeploy/vision/ocr/model.cs rename to bindings/csharp/fastdeploy/vision/ocr/model.cs diff --git a/csharp/fastdeploy/vision/result.cs b/bindings/csharp/fastdeploy/vision/result.cs similarity index 100% rename from csharp/fastdeploy/vision/result.cs rename to bindings/csharp/fastdeploy/vision/result.cs diff --git a/csharp/fastdeploy/vision/segmentation/model.cs b/bindings/csharp/fastdeploy/vision/segmentation/model.cs similarity index 100% rename from csharp/fastdeploy/vision/segmentation/model.cs rename to bindings/csharp/fastdeploy/vision/segmentation/model.cs diff --git a/csharp/fastdeploy/vision/visualize.cs b/bindings/csharp/fastdeploy/vision/visualize.cs similarity index 100% rename from csharp/fastdeploy/vision/visualize.cs rename to bindings/csharp/fastdeploy/vision/visualize.cs diff --git a/java/android/.gitignore b/bindings/java/android/.gitignore similarity index 100% rename from java/android/.gitignore rename to bindings/java/android/.gitignore diff --git a/java/android/README.md b/bindings/java/android/README.md similarity index 100% rename from java/android/README.md rename to bindings/java/android/README.md diff --git a/java/android/README_CN.md b/bindings/java/android/README_CN.md similarity index 100% rename from java/android/README_CN.md rename to bindings/java/android/README_CN.md diff --git a/java/android/app/build.gradle b/bindings/java/android/app/build.gradle similarity index 100% rename from java/android/app/build.gradle rename to bindings/java/android/app/build.gradle diff --git a/java/android/app/proguard-rules.pro b/bindings/java/android/app/proguard-rules.pro similarity index 100% rename from java/android/app/proguard-rules.pro rename to bindings/java/android/app/proguard-rules.pro diff --git a/java/android/app/src/main/AndroidManifest.xml b/bindings/java/android/app/src/main/AndroidManifest.xml similarity index 100% rename from java/android/app/src/main/AndroidManifest.xml rename to bindings/java/android/app/src/main/AndroidManifest.xml diff --git a/java/android/app/src/main/assets/labels/coco_label_list.txt b/bindings/java/android/app/src/main/assets/labels/coco_label_list.txt similarity index 100% rename from java/android/app/src/main/assets/labels/coco_label_list.txt rename to bindings/java/android/app/src/main/assets/labels/coco_label_list.txt diff --git a/java/android/app/src/main/assets/labels/en_dict.txt b/bindings/java/android/app/src/main/assets/labels/en_dict.txt similarity index 100% rename from java/android/app/src/main/assets/labels/en_dict.txt rename to bindings/java/android/app/src/main/assets/labels/en_dict.txt diff --git a/java/android/app/src/main/assets/labels/imagenet1k_label_list.txt b/bindings/java/android/app/src/main/assets/labels/imagenet1k_label_list.txt similarity index 100% rename from java/android/app/src/main/assets/labels/imagenet1k_label_list.txt rename to bindings/java/android/app/src/main/assets/labels/imagenet1k_label_list.txt diff --git a/java/android/app/src/main/assets/labels/pascalvoc_label_list b/bindings/java/android/app/src/main/assets/labels/pascalvoc_label_list similarity index 100% rename from java/android/app/src/main/assets/labels/pascalvoc_label_list rename to bindings/java/android/app/src/main/assets/labels/pascalvoc_label_list diff --git a/java/android/app/src/main/assets/labels/ppocr_keys_v1.txt b/bindings/java/android/app/src/main/assets/labels/ppocr_keys_v1.txt similarity index 100% rename from java/android/app/src/main/assets/labels/ppocr_keys_v1.txt rename to bindings/java/android/app/src/main/assets/labels/ppocr_keys_v1.txt diff --git a/java/android/app/src/main/assets/super_pic_-2.jpg b/bindings/java/android/app/src/main/assets/super_pic_-2.jpg similarity index 100% rename from java/android/app/src/main/assets/super_pic_-2.jpg rename to bindings/java/android/app/src/main/assets/super_pic_-2.jpg diff --git a/java/android/app/src/main/assets/super_pic_1.jpg b/bindings/java/android/app/src/main/assets/super_pic_1.jpg similarity index 100% rename from java/android/app/src/main/assets/super_pic_1.jpg rename to bindings/java/android/app/src/main/assets/super_pic_1.jpg diff --git a/java/android/app/src/main/assets/super_pic_2.jpg b/bindings/java/android/app/src/main/assets/super_pic_2.jpg similarity index 100% rename from java/android/app/src/main/assets/super_pic_2.jpg rename to bindings/java/android/app/src/main/assets/super_pic_2.jpg diff --git a/java/android/app/src/main/assets/super_pic_4.jpg b/bindings/java/android/app/src/main/assets/super_pic_4.jpg similarity index 100% rename from java/android/app/src/main/assets/super_pic_4.jpg rename to bindings/java/android/app/src/main/assets/super_pic_4.jpg diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/classification/ClassificationMainActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/classification/ClassificationMainActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/classification/ClassificationMainActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/classification/ClassificationMainActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/classification/ClassificationSettingsActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/classification/ClassificationSettingsActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/classification/ClassificationSettingsActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/classification/ClassificationSettingsActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/classification/ClassificationWelcomeActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/classification/ClassificationWelcomeActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/classification/ClassificationWelcomeActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/classification/ClassificationWelcomeActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/detection/DetectionMainActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/detection/DetectionMainActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/detection/DetectionMainActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/detection/DetectionMainActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/detection/DetectionSettingsActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/detection/DetectionSettingsActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/detection/DetectionSettingsActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/detection/DetectionSettingsActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/detection/DetectionWelcomeActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/detection/DetectionWelcomeActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/detection/DetectionWelcomeActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/detection/DetectionWelcomeActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facealign/FaceAlignMainActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facealign/FaceAlignMainActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facealign/FaceAlignMainActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facealign/FaceAlignMainActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facealign/FaceAlignSettingsActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facealign/FaceAlignSettingsActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facealign/FaceAlignSettingsActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facealign/FaceAlignSettingsActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facealign/FaceAlignWelcomeActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facealign/FaceAlignWelcomeActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facealign/FaceAlignWelcomeActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facealign/FaceAlignWelcomeActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facedet/FaceDetMainActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facedet/FaceDetMainActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facedet/FaceDetMainActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facedet/FaceDetMainActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facedet/FaceDetSettingsActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facedet/FaceDetSettingsActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facedet/FaceDetSettingsActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facedet/FaceDetSettingsActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facedet/FaceDetWelcomeActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facedet/FaceDetWelcomeActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facedet/FaceDetWelcomeActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/facedet/FaceDetWelcomeActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/keypointdetection/KeyPointDetectionMainActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/keypointdetection/KeyPointDetectionMainActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/keypointdetection/KeyPointDetectionMainActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/keypointdetection/KeyPointDetectionMainActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/keypointdetection/KeyPointDetectionSettingsActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/keypointdetection/KeyPointDetectionSettingsActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/keypointdetection/KeyPointDetectionSettingsActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/keypointdetection/KeyPointDetectionSettingsActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/keypointdetection/KeyPointDetectionWelcomeActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/keypointdetection/KeyPointDetectionWelcomeActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/keypointdetection/KeyPointDetectionWelcomeActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/keypointdetection/KeyPointDetectionWelcomeActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/matting/MattingMainActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/matting/MattingMainActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/matting/MattingMainActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/matting/MattingMainActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/matting/MattingSettingsActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/matting/MattingSettingsActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/matting/MattingSettingsActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/matting/MattingSettingsActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/matting/MattingWelcomeActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/matting/MattingWelcomeActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/matting/MattingWelcomeActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/matting/MattingWelcomeActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/ocr/OcrMainActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/ocr/OcrMainActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/ocr/OcrMainActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/ocr/OcrMainActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/ocr/OcrSettingsActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/ocr/OcrSettingsActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/ocr/OcrSettingsActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/ocr/OcrSettingsActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/ocr/OcrWelcomeActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/ocr/OcrWelcomeActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/ocr/OcrWelcomeActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/ocr/OcrWelcomeActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/segmentation/SegmentationMainActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/segmentation/SegmentationMainActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/segmentation/SegmentationMainActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/segmentation/SegmentationMainActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/segmentation/SegmentationSettingsActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/segmentation/SegmentationSettingsActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/segmentation/SegmentationSettingsActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/segmentation/SegmentationSettingsActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/segmentation/SegmentationWelcomeActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/segmentation/SegmentationWelcomeActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/segmentation/SegmentationWelcomeActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/segmentation/SegmentationWelcomeActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/sr/SuperResolutionMainActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/sr/SuperResolutionMainActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/sr/SuperResolutionMainActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/sr/SuperResolutionMainActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/sr/SuperResolutionSettingsActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/sr/SuperResolutionSettingsActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/sr/SuperResolutionSettingsActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/sr/SuperResolutionSettingsActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/sr/SuperResolutionWelcomeActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/sr/SuperResolutionWelcomeActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/sr/SuperResolutionWelcomeActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/sr/SuperResolutionWelcomeActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/applications/VoiceAssistantMainActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/applications/VoiceAssistantMainActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/applications/VoiceAssistantMainActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/applications/VoiceAssistantMainActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/applications/VoiceAssistantSettingsActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/applications/VoiceAssistantSettingsActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/applications/VoiceAssistantSettingsActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/applications/VoiceAssistantSettingsActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/applications/VoiceAssistantWelcomeActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/applications/VoiceAssistantWelcomeActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/applications/VoiceAssistantWelcomeActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/applications/VoiceAssistantWelcomeActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/uie/UIEMainActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/uie/UIEMainActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/uie/UIEMainActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/uie/UIEMainActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/uie/UIESettingsActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/uie/UIESettingsActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/uie/UIESettingsActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/uie/UIESettingsActivity.java diff --git a/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/uie/UIEWelcomeActivity.java b/bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/uie/UIEWelcomeActivity.java similarity index 100% rename from java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/uie/UIEWelcomeActivity.java rename to bindings/java/android/app/src/main/java/com/baidu/paddle/fastdeploy/app/examples/text/uie/UIEWelcomeActivity.java diff --git a/java/android/app/src/main/res/drawable-v24/action_button_layer.xml b/bindings/java/android/app/src/main/res/drawable-v24/action_button_layer.xml similarity index 100% rename from java/android/app/src/main/res/drawable-v24/action_button_layer.xml rename to bindings/java/android/app/src/main/res/drawable-v24/action_button_layer.xml diff --git a/java/android/app/src/main/res/drawable-v24/album_btn.xml b/bindings/java/android/app/src/main/res/drawable-v24/album_btn.xml similarity index 100% rename from java/android/app/src/main/res/drawable-v24/album_btn.xml rename to bindings/java/android/app/src/main/res/drawable-v24/album_btn.xml diff --git a/java/android/app/src/main/res/drawable-v24/ic_launcher_foreground.xml b/bindings/java/android/app/src/main/res/drawable-v24/ic_launcher_foreground.xml similarity index 100% rename from java/android/app/src/main/res/drawable-v24/ic_launcher_foreground.xml rename to bindings/java/android/app/src/main/res/drawable-v24/ic_launcher_foreground.xml diff --git a/java/android/app/src/main/res/drawable-v24/realtime_start_btn.xml b/bindings/java/android/app/src/main/res/drawable-v24/realtime_start_btn.xml similarity index 100% rename from java/android/app/src/main/res/drawable-v24/realtime_start_btn.xml rename to bindings/java/android/app/src/main/res/drawable-v24/realtime_start_btn.xml diff --git a/java/android/app/src/main/res/drawable-v24/realtime_stop_btn.xml b/bindings/java/android/app/src/main/res/drawable-v24/realtime_stop_btn.xml similarity index 100% rename from java/android/app/src/main/res/drawable-v24/realtime_stop_btn.xml rename to bindings/java/android/app/src/main/res/drawable-v24/realtime_stop_btn.xml diff --git a/java/android/app/src/main/res/drawable-v24/result_page_border_section_bk.xml b/bindings/java/android/app/src/main/res/drawable-v24/result_page_border_section_bk.xml similarity index 100% rename from java/android/app/src/main/res/drawable-v24/result_page_border_section_bk.xml rename to bindings/java/android/app/src/main/res/drawable-v24/result_page_border_section_bk.xml diff --git a/java/android/app/src/main/res/drawable-v24/round_corner_btn.xml b/bindings/java/android/app/src/main/res/drawable-v24/round_corner_btn.xml similarity index 100% rename from java/android/app/src/main/res/drawable-v24/round_corner_btn.xml rename to bindings/java/android/app/src/main/res/drawable-v24/round_corner_btn.xml diff --git a/java/android/app/src/main/res/drawable-v24/seekbar_progress_realtime.xml b/bindings/java/android/app/src/main/res/drawable-v24/seekbar_progress_realtime.xml similarity index 100% rename from java/android/app/src/main/res/drawable-v24/seekbar_progress_realtime.xml rename to bindings/java/android/app/src/main/res/drawable-v24/seekbar_progress_realtime.xml diff --git a/java/android/app/src/main/res/drawable-v24/seekbar_progress_result.xml b/bindings/java/android/app/src/main/res/drawable-v24/seekbar_progress_result.xml similarity index 100% rename from java/android/app/src/main/res/drawable-v24/seekbar_progress_result.xml rename to bindings/java/android/app/src/main/res/drawable-v24/seekbar_progress_result.xml diff --git a/java/android/app/src/main/res/drawable-v24/seekbar_thumb.xml b/bindings/java/android/app/src/main/res/drawable-v24/seekbar_thumb.xml similarity index 100% rename from java/android/app/src/main/res/drawable-v24/seekbar_thumb.xml rename to bindings/java/android/app/src/main/res/drawable-v24/seekbar_thumb.xml diff --git a/java/android/app/src/main/res/drawable-v24/seekbar_thumb_shape.xml b/bindings/java/android/app/src/main/res/drawable-v24/seekbar_thumb_shape.xml similarity index 100% rename from java/android/app/src/main/res/drawable-v24/seekbar_thumb_shape.xml rename to bindings/java/android/app/src/main/res/drawable-v24/seekbar_thumb_shape.xml diff --git a/java/android/app/src/main/res/drawable-v24/switch_side_btn.xml b/bindings/java/android/app/src/main/res/drawable-v24/switch_side_btn.xml similarity index 100% rename from java/android/app/src/main/res/drawable-v24/switch_side_btn.xml rename to bindings/java/android/app/src/main/res/drawable-v24/switch_side_btn.xml diff --git a/java/android/app/src/main/res/drawable-v24/take_picture_btn.xml b/bindings/java/android/app/src/main/res/drawable-v24/take_picture_btn.xml similarity index 100% rename from java/android/app/src/main/res/drawable-v24/take_picture_btn.xml rename to bindings/java/android/app/src/main/res/drawable-v24/take_picture_btn.xml diff --git a/java/android/app/src/main/res/drawable-xhdpi/album.png b/bindings/java/android/app/src/main/res/drawable-xhdpi/album.png similarity index 100% rename from java/android/app/src/main/res/drawable-xhdpi/album.png rename to bindings/java/android/app/src/main/res/drawable-xhdpi/album.png diff --git a/java/android/app/src/main/res/drawable-xhdpi/album_pressed.png b/bindings/java/android/app/src/main/res/drawable-xhdpi/album_pressed.png similarity index 100% rename from java/android/app/src/main/res/drawable-xhdpi/album_pressed.png rename to bindings/java/android/app/src/main/res/drawable-xhdpi/album_pressed.png diff --git a/java/android/app/src/main/res/drawable-xhdpi/back_btn.png b/bindings/java/android/app/src/main/res/drawable-xhdpi/back_btn.png similarity index 100% rename from java/android/app/src/main/res/drawable-xhdpi/back_btn.png rename to bindings/java/android/app/src/main/res/drawable-xhdpi/back_btn.png diff --git a/java/android/app/src/main/res/drawable-xhdpi/more_menu.png b/bindings/java/android/app/src/main/res/drawable-xhdpi/more_menu.png similarity index 100% rename from java/android/app/src/main/res/drawable-xhdpi/more_menu.png rename to bindings/java/android/app/src/main/res/drawable-xhdpi/more_menu.png diff --git a/java/android/app/src/main/res/drawable-xhdpi/realtime_start.png b/bindings/java/android/app/src/main/res/drawable-xhdpi/realtime_start.png similarity index 100% rename from java/android/app/src/main/res/drawable-xhdpi/realtime_start.png rename to bindings/java/android/app/src/main/res/drawable-xhdpi/realtime_start.png diff --git a/java/android/app/src/main/res/drawable-xhdpi/realtime_start_pressed.png b/bindings/java/android/app/src/main/res/drawable-xhdpi/realtime_start_pressed.png similarity index 100% rename from java/android/app/src/main/res/drawable-xhdpi/realtime_start_pressed.png rename to bindings/java/android/app/src/main/res/drawable-xhdpi/realtime_start_pressed.png diff --git a/java/android/app/src/main/res/drawable-xhdpi/realtime_stop.png b/bindings/java/android/app/src/main/res/drawable-xhdpi/realtime_stop.png similarity index 100% rename from java/android/app/src/main/res/drawable-xhdpi/realtime_stop.png rename to bindings/java/android/app/src/main/res/drawable-xhdpi/realtime_stop.png diff --git a/java/android/app/src/main/res/drawable-xhdpi/realtime_stop_pressed.png b/bindings/java/android/app/src/main/res/drawable-xhdpi/realtime_stop_pressed.png similarity index 100% rename from java/android/app/src/main/res/drawable-xhdpi/realtime_stop_pressed.png rename to bindings/java/android/app/src/main/res/drawable-xhdpi/realtime_stop_pressed.png diff --git a/java/android/app/src/main/res/drawable-xhdpi/scan_icon.png b/bindings/java/android/app/src/main/res/drawable-xhdpi/scan_icon.png similarity index 100% rename from java/android/app/src/main/res/drawable-xhdpi/scan_icon.png rename to bindings/java/android/app/src/main/res/drawable-xhdpi/scan_icon.png diff --git a/java/android/app/src/main/res/drawable-xhdpi/seekbar_handle.png b/bindings/java/android/app/src/main/res/drawable-xhdpi/seekbar_handle.png similarity index 100% rename from java/android/app/src/main/res/drawable-xhdpi/seekbar_handle.png rename to bindings/java/android/app/src/main/res/drawable-xhdpi/seekbar_handle.png diff --git a/java/android/app/src/main/res/drawable-xhdpi/seekbar_progress_dotted.png b/bindings/java/android/app/src/main/res/drawable-xhdpi/seekbar_progress_dotted.png similarity index 100% rename from java/android/app/src/main/res/drawable-xhdpi/seekbar_progress_dotted.png rename to bindings/java/android/app/src/main/res/drawable-xhdpi/seekbar_progress_dotted.png diff --git a/java/android/app/src/main/res/drawable-xhdpi/seekbar_thumb_invisible.png b/bindings/java/android/app/src/main/res/drawable-xhdpi/seekbar_thumb_invisible.png similarity index 100% rename from java/android/app/src/main/res/drawable-xhdpi/seekbar_thumb_invisible.png rename to bindings/java/android/app/src/main/res/drawable-xhdpi/seekbar_thumb_invisible.png diff --git a/java/android/app/src/main/res/drawable-xhdpi/switch_side.png b/bindings/java/android/app/src/main/res/drawable-xhdpi/switch_side.png similarity index 100% rename from java/android/app/src/main/res/drawable-xhdpi/switch_side.png rename to bindings/java/android/app/src/main/res/drawable-xhdpi/switch_side.png diff --git a/java/android/app/src/main/res/drawable-xhdpi/switch_side_pressed.png b/bindings/java/android/app/src/main/res/drawable-xhdpi/switch_side_pressed.png similarity index 100% rename from java/android/app/src/main/res/drawable-xhdpi/switch_side_pressed.png rename to bindings/java/android/app/src/main/res/drawable-xhdpi/switch_side_pressed.png diff --git a/java/android/app/src/main/res/drawable-xhdpi/take_picture.png b/bindings/java/android/app/src/main/res/drawable-xhdpi/take_picture.png similarity index 100% rename from java/android/app/src/main/res/drawable-xhdpi/take_picture.png rename to bindings/java/android/app/src/main/res/drawable-xhdpi/take_picture.png diff --git a/java/android/app/src/main/res/drawable-xhdpi/take_picture_pressed.png b/bindings/java/android/app/src/main/res/drawable-xhdpi/take_picture_pressed.png similarity index 100% rename from java/android/app/src/main/res/drawable-xhdpi/take_picture_pressed.png rename to bindings/java/android/app/src/main/res/drawable-xhdpi/take_picture_pressed.png diff --git a/java/android/app/src/main/res/drawable-xxhdpi-v4/btn_switch_default.png b/bindings/java/android/app/src/main/res/drawable-xxhdpi-v4/btn_switch_default.png similarity index 100% rename from java/android/app/src/main/res/drawable-xxhdpi-v4/btn_switch_default.png rename to bindings/java/android/app/src/main/res/drawable-xxhdpi-v4/btn_switch_default.png diff --git a/java/android/app/src/main/res/drawable-xxhdpi-v4/btn_switch_pressed.png b/bindings/java/android/app/src/main/res/drawable-xxhdpi-v4/btn_switch_pressed.png similarity index 100% rename from java/android/app/src/main/res/drawable-xxhdpi-v4/btn_switch_pressed.png rename to bindings/java/android/app/src/main/res/drawable-xxhdpi-v4/btn_switch_pressed.png diff --git a/java/android/app/src/main/res/drawable/btn_settings.xml b/bindings/java/android/app/src/main/res/drawable/btn_settings.xml similarity index 100% rename from java/android/app/src/main/res/drawable/btn_settings.xml rename to bindings/java/android/app/src/main/res/drawable/btn_settings.xml diff --git a/java/android/app/src/main/res/drawable/btn_settings_default.xml b/bindings/java/android/app/src/main/res/drawable/btn_settings_default.xml similarity index 100% rename from java/android/app/src/main/res/drawable/btn_settings_default.xml rename to bindings/java/android/app/src/main/res/drawable/btn_settings_default.xml diff --git a/java/android/app/src/main/res/drawable/btn_settings_pressed.xml b/bindings/java/android/app/src/main/res/drawable/btn_settings_pressed.xml similarity index 100% rename from java/android/app/src/main/res/drawable/btn_settings_pressed.xml rename to bindings/java/android/app/src/main/res/drawable/btn_settings_pressed.xml diff --git a/java/android/app/src/main/res/drawable/btn_shutter.xml b/bindings/java/android/app/src/main/res/drawable/btn_shutter.xml similarity index 100% rename from java/android/app/src/main/res/drawable/btn_shutter.xml rename to bindings/java/android/app/src/main/res/drawable/btn_shutter.xml diff --git a/java/android/app/src/main/res/drawable/btn_shutter_default.xml b/bindings/java/android/app/src/main/res/drawable/btn_shutter_default.xml similarity index 100% rename from java/android/app/src/main/res/drawable/btn_shutter_default.xml rename to bindings/java/android/app/src/main/res/drawable/btn_shutter_default.xml diff --git a/java/android/app/src/main/res/drawable/btn_shutter_pressed.xml b/bindings/java/android/app/src/main/res/drawable/btn_shutter_pressed.xml similarity index 100% rename from java/android/app/src/main/res/drawable/btn_shutter_pressed.xml rename to bindings/java/android/app/src/main/res/drawable/btn_shutter_pressed.xml diff --git a/java/android/app/src/main/res/drawable/btn_switch.xml b/bindings/java/android/app/src/main/res/drawable/btn_switch.xml similarity index 100% rename from java/android/app/src/main/res/drawable/btn_switch.xml rename to bindings/java/android/app/src/main/res/drawable/btn_switch.xml diff --git a/java/android/app/src/main/res/drawable/ic_launcher_background.xml b/bindings/java/android/app/src/main/res/drawable/ic_launcher_background.xml similarity index 100% rename from java/android/app/src/main/res/drawable/ic_launcher_background.xml rename to bindings/java/android/app/src/main/res/drawable/ic_launcher_background.xml diff --git a/java/android/app/src/main/res/drawable/main_bk.png b/bindings/java/android/app/src/main/res/drawable/main_bk.png similarity index 100% rename from java/android/app/src/main/res/drawable/main_bk.png rename to bindings/java/android/app/src/main/res/drawable/main_bk.png diff --git a/java/android/app/src/main/res/drawable/paddle_logo.png b/bindings/java/android/app/src/main/res/drawable/paddle_logo.png similarity index 100% rename from java/android/app/src/main/res/drawable/paddle_logo.png rename to bindings/java/android/app/src/main/res/drawable/paddle_logo.png diff --git a/java/android/app/src/main/res/layout-land/detection_activity_main.xml b/bindings/java/android/app/src/main/res/layout-land/detection_activity_main.xml similarity index 100% rename from java/android/app/src/main/res/layout-land/detection_activity_main.xml rename to bindings/java/android/app/src/main/res/layout-land/detection_activity_main.xml diff --git a/java/android/app/src/main/res/layout-land/ocr_activity_main.xml b/bindings/java/android/app/src/main/res/layout-land/ocr_activity_main.xml similarity index 100% rename from java/android/app/src/main/res/layout-land/ocr_activity_main.xml rename to bindings/java/android/app/src/main/res/layout-land/ocr_activity_main.xml diff --git a/java/android/app/src/main/res/layout/classification_activity_main.xml b/bindings/java/android/app/src/main/res/layout/classification_activity_main.xml similarity index 100% rename from java/android/app/src/main/res/layout/classification_activity_main.xml rename to bindings/java/android/app/src/main/res/layout/classification_activity_main.xml diff --git a/java/android/app/src/main/res/layout/classification_camera_page.xml b/bindings/java/android/app/src/main/res/layout/classification_camera_page.xml similarity index 100% rename from java/android/app/src/main/res/layout/classification_camera_page.xml rename to bindings/java/android/app/src/main/res/layout/classification_camera_page.xml diff --git a/java/android/app/src/main/res/layout/classification_result_page.xml b/bindings/java/android/app/src/main/res/layout/classification_result_page.xml similarity index 100% rename from java/android/app/src/main/res/layout/classification_result_page.xml rename to bindings/java/android/app/src/main/res/layout/classification_result_page.xml diff --git a/java/android/app/src/main/res/layout/classification_welcome.xml b/bindings/java/android/app/src/main/res/layout/classification_welcome.xml similarity index 100% rename from java/android/app/src/main/res/layout/classification_welcome.xml rename to bindings/java/android/app/src/main/res/layout/classification_welcome.xml diff --git a/java/android/app/src/main/res/layout/detection_activity_main.xml b/bindings/java/android/app/src/main/res/layout/detection_activity_main.xml similarity index 100% rename from java/android/app/src/main/res/layout/detection_activity_main.xml rename to bindings/java/android/app/src/main/res/layout/detection_activity_main.xml diff --git a/java/android/app/src/main/res/layout/detection_camera_page.xml b/bindings/java/android/app/src/main/res/layout/detection_camera_page.xml similarity index 100% rename from java/android/app/src/main/res/layout/detection_camera_page.xml rename to bindings/java/android/app/src/main/res/layout/detection_camera_page.xml diff --git a/java/android/app/src/main/res/layout/detection_result_page.xml b/bindings/java/android/app/src/main/res/layout/detection_result_page.xml similarity index 100% rename from java/android/app/src/main/res/layout/detection_result_page.xml rename to bindings/java/android/app/src/main/res/layout/detection_result_page.xml diff --git a/java/android/app/src/main/res/layout/detection_welcome.xml b/bindings/java/android/app/src/main/res/layout/detection_welcome.xml similarity index 100% rename from java/android/app/src/main/res/layout/detection_welcome.xml rename to bindings/java/android/app/src/main/res/layout/detection_welcome.xml diff --git a/java/android/app/src/main/res/layout/face_align_activity_main.xml b/bindings/java/android/app/src/main/res/layout/face_align_activity_main.xml similarity index 100% rename from java/android/app/src/main/res/layout/face_align_activity_main.xml rename to bindings/java/android/app/src/main/res/layout/face_align_activity_main.xml diff --git a/java/android/app/src/main/res/layout/face_align_camera_page.xml b/bindings/java/android/app/src/main/res/layout/face_align_camera_page.xml similarity index 100% rename from java/android/app/src/main/res/layout/face_align_camera_page.xml rename to bindings/java/android/app/src/main/res/layout/face_align_camera_page.xml diff --git a/java/android/app/src/main/res/layout/face_align_result_page.xml b/bindings/java/android/app/src/main/res/layout/face_align_result_page.xml similarity index 100% rename from java/android/app/src/main/res/layout/face_align_result_page.xml rename to bindings/java/android/app/src/main/res/layout/face_align_result_page.xml diff --git a/java/android/app/src/main/res/layout/face_align_welcome.xml b/bindings/java/android/app/src/main/res/layout/face_align_welcome.xml similarity index 100% rename from java/android/app/src/main/res/layout/face_align_welcome.xml rename to bindings/java/android/app/src/main/res/layout/face_align_welcome.xml diff --git a/java/android/app/src/main/res/layout/facedet_activity_main.xml b/bindings/java/android/app/src/main/res/layout/facedet_activity_main.xml similarity index 100% rename from java/android/app/src/main/res/layout/facedet_activity_main.xml rename to bindings/java/android/app/src/main/res/layout/facedet_activity_main.xml diff --git a/java/android/app/src/main/res/layout/facedet_camera_page.xml b/bindings/java/android/app/src/main/res/layout/facedet_camera_page.xml similarity index 100% rename from java/android/app/src/main/res/layout/facedet_camera_page.xml rename to bindings/java/android/app/src/main/res/layout/facedet_camera_page.xml diff --git a/java/android/app/src/main/res/layout/facedet_result_page.xml b/bindings/java/android/app/src/main/res/layout/facedet_result_page.xml similarity index 100% rename from java/android/app/src/main/res/layout/facedet_result_page.xml rename to bindings/java/android/app/src/main/res/layout/facedet_result_page.xml diff --git a/java/android/app/src/main/res/layout/facedet_welcome.xml b/bindings/java/android/app/src/main/res/layout/facedet_welcome.xml similarity index 100% rename from java/android/app/src/main/res/layout/facedet_welcome.xml rename to bindings/java/android/app/src/main/res/layout/facedet_welcome.xml diff --git a/java/android/app/src/main/res/layout/keypointdetection_activity_main.xml b/bindings/java/android/app/src/main/res/layout/keypointdetection_activity_main.xml similarity index 100% rename from java/android/app/src/main/res/layout/keypointdetection_activity_main.xml rename to bindings/java/android/app/src/main/res/layout/keypointdetection_activity_main.xml diff --git a/java/android/app/src/main/res/layout/keypointdetection_camera_page.xml b/bindings/java/android/app/src/main/res/layout/keypointdetection_camera_page.xml similarity index 100% rename from java/android/app/src/main/res/layout/keypointdetection_camera_page.xml rename to bindings/java/android/app/src/main/res/layout/keypointdetection_camera_page.xml diff --git a/java/android/app/src/main/res/layout/keypointdetection_result_page.xml b/bindings/java/android/app/src/main/res/layout/keypointdetection_result_page.xml similarity index 100% rename from java/android/app/src/main/res/layout/keypointdetection_result_page.xml rename to bindings/java/android/app/src/main/res/layout/keypointdetection_result_page.xml diff --git a/java/android/app/src/main/res/layout/keypointdetection_welcome.xml b/bindings/java/android/app/src/main/res/layout/keypointdetection_welcome.xml similarity index 100% rename from java/android/app/src/main/res/layout/keypointdetection_welcome.xml rename to bindings/java/android/app/src/main/res/layout/keypointdetection_welcome.xml diff --git a/java/android/app/src/main/res/layout/matting_activity_main.xml b/bindings/java/android/app/src/main/res/layout/matting_activity_main.xml similarity index 100% rename from java/android/app/src/main/res/layout/matting_activity_main.xml rename to bindings/java/android/app/src/main/res/layout/matting_activity_main.xml diff --git a/java/android/app/src/main/res/layout/matting_camera_page.xml b/bindings/java/android/app/src/main/res/layout/matting_camera_page.xml similarity index 100% rename from java/android/app/src/main/res/layout/matting_camera_page.xml rename to bindings/java/android/app/src/main/res/layout/matting_camera_page.xml diff --git a/java/android/app/src/main/res/layout/matting_result_page.xml b/bindings/java/android/app/src/main/res/layout/matting_result_page.xml similarity index 100% rename from java/android/app/src/main/res/layout/matting_result_page.xml rename to bindings/java/android/app/src/main/res/layout/matting_result_page.xml diff --git a/java/android/app/src/main/res/layout/matting_welcome.xml b/bindings/java/android/app/src/main/res/layout/matting_welcome.xml similarity index 100% rename from java/android/app/src/main/res/layout/matting_welcome.xml rename to bindings/java/android/app/src/main/res/layout/matting_welcome.xml diff --git a/java/android/app/src/main/res/layout/ocr_activity_main.xml b/bindings/java/android/app/src/main/res/layout/ocr_activity_main.xml similarity index 100% rename from java/android/app/src/main/res/layout/ocr_activity_main.xml rename to bindings/java/android/app/src/main/res/layout/ocr_activity_main.xml diff --git a/java/android/app/src/main/res/layout/ocr_camera_page.xml b/bindings/java/android/app/src/main/res/layout/ocr_camera_page.xml similarity index 100% rename from java/android/app/src/main/res/layout/ocr_camera_page.xml rename to bindings/java/android/app/src/main/res/layout/ocr_camera_page.xml diff --git a/java/android/app/src/main/res/layout/ocr_result_page.xml b/bindings/java/android/app/src/main/res/layout/ocr_result_page.xml similarity index 100% rename from java/android/app/src/main/res/layout/ocr_result_page.xml rename to bindings/java/android/app/src/main/res/layout/ocr_result_page.xml diff --git a/java/android/app/src/main/res/layout/ocr_welcome.xml b/bindings/java/android/app/src/main/res/layout/ocr_welcome.xml similarity index 100% rename from java/android/app/src/main/res/layout/ocr_welcome.xml rename to bindings/java/android/app/src/main/res/layout/ocr_welcome.xml diff --git a/java/android/app/src/main/res/layout/segmentation_activity_main.xml b/bindings/java/android/app/src/main/res/layout/segmentation_activity_main.xml similarity index 100% rename from java/android/app/src/main/res/layout/segmentation_activity_main.xml rename to bindings/java/android/app/src/main/res/layout/segmentation_activity_main.xml diff --git a/java/android/app/src/main/res/layout/segmentation_camera_page.xml b/bindings/java/android/app/src/main/res/layout/segmentation_camera_page.xml similarity index 100% rename from java/android/app/src/main/res/layout/segmentation_camera_page.xml rename to bindings/java/android/app/src/main/res/layout/segmentation_camera_page.xml diff --git a/java/android/app/src/main/res/layout/segmentation_result_page.xml b/bindings/java/android/app/src/main/res/layout/segmentation_result_page.xml similarity index 100% rename from java/android/app/src/main/res/layout/segmentation_result_page.xml rename to bindings/java/android/app/src/main/res/layout/segmentation_result_page.xml diff --git a/java/android/app/src/main/res/layout/segmentation_welcome.xml b/bindings/java/android/app/src/main/res/layout/segmentation_welcome.xml similarity index 100% rename from java/android/app/src/main/res/layout/segmentation_welcome.xml rename to bindings/java/android/app/src/main/res/layout/segmentation_welcome.xml diff --git a/java/android/app/src/main/res/layout/super_resolution_activity_main.xml b/bindings/java/android/app/src/main/res/layout/super_resolution_activity_main.xml similarity index 100% rename from java/android/app/src/main/res/layout/super_resolution_activity_main.xml rename to bindings/java/android/app/src/main/res/layout/super_resolution_activity_main.xml diff --git a/java/android/app/src/main/res/layout/super_resolution_welcome.xml b/bindings/java/android/app/src/main/res/layout/super_resolution_welcome.xml similarity index 100% rename from java/android/app/src/main/res/layout/super_resolution_welcome.xml rename to bindings/java/android/app/src/main/res/layout/super_resolution_welcome.xml diff --git a/java/android/app/src/main/res/layout/uie_activity_main.xml b/bindings/java/android/app/src/main/res/layout/uie_activity_main.xml similarity index 100% rename from java/android/app/src/main/res/layout/uie_activity_main.xml rename to bindings/java/android/app/src/main/res/layout/uie_activity_main.xml diff --git a/java/android/app/src/main/res/layout/uie_welcome.xml b/bindings/java/android/app/src/main/res/layout/uie_welcome.xml similarity index 100% rename from java/android/app/src/main/res/layout/uie_welcome.xml rename to bindings/java/android/app/src/main/res/layout/uie_welcome.xml diff --git a/java/android/app/src/main/res/layout/voice_assistant_activity_main.xml b/bindings/java/android/app/src/main/res/layout/voice_assistant_activity_main.xml similarity index 100% rename from java/android/app/src/main/res/layout/voice_assistant_activity_main.xml rename to bindings/java/android/app/src/main/res/layout/voice_assistant_activity_main.xml diff --git a/java/android/app/src/main/res/layout/voice_assistant_welcome.xml b/bindings/java/android/app/src/main/res/layout/voice_assistant_welcome.xml similarity index 100% rename from java/android/app/src/main/res/layout/voice_assistant_welcome.xml rename to bindings/java/android/app/src/main/res/layout/voice_assistant_welcome.xml diff --git a/java/android/app/src/main/res/mipmap-anydpi-v26/ic_launcher.xml b/bindings/java/android/app/src/main/res/mipmap-anydpi-v26/ic_launcher.xml similarity index 100% rename from java/android/app/src/main/res/mipmap-anydpi-v26/ic_launcher.xml rename to bindings/java/android/app/src/main/res/mipmap-anydpi-v26/ic_launcher.xml diff --git a/java/android/app/src/main/res/mipmap-anydpi-v26/ic_launcher_round.xml b/bindings/java/android/app/src/main/res/mipmap-anydpi-v26/ic_launcher_round.xml similarity index 100% rename from java/android/app/src/main/res/mipmap-anydpi-v26/ic_launcher_round.xml rename to bindings/java/android/app/src/main/res/mipmap-anydpi-v26/ic_launcher_round.xml diff --git a/java/android/app/src/main/res/mipmap-hdpi/ic_launcher.png b/bindings/java/android/app/src/main/res/mipmap-hdpi/ic_launcher.png similarity index 100% rename from java/android/app/src/main/res/mipmap-hdpi/ic_launcher.png rename to bindings/java/android/app/src/main/res/mipmap-hdpi/ic_launcher.png diff --git a/java/android/app/src/main/res/mipmap-hdpi/ic_launcher_round.png b/bindings/java/android/app/src/main/res/mipmap-hdpi/ic_launcher_round.png similarity index 100% rename from java/android/app/src/main/res/mipmap-hdpi/ic_launcher_round.png rename to bindings/java/android/app/src/main/res/mipmap-hdpi/ic_launcher_round.png diff --git a/java/android/app/src/main/res/mipmap-mdpi/ic_launcher.png b/bindings/java/android/app/src/main/res/mipmap-mdpi/ic_launcher.png similarity index 100% rename from java/android/app/src/main/res/mipmap-mdpi/ic_launcher.png rename to bindings/java/android/app/src/main/res/mipmap-mdpi/ic_launcher.png diff --git a/java/android/app/src/main/res/mipmap-mdpi/ic_launcher_round.png b/bindings/java/android/app/src/main/res/mipmap-mdpi/ic_launcher_round.png similarity index 100% rename from java/android/app/src/main/res/mipmap-mdpi/ic_launcher_round.png rename to bindings/java/android/app/src/main/res/mipmap-mdpi/ic_launcher_round.png diff --git a/java/android/app/src/main/res/mipmap-xhdpi/ic_launcher.png b/bindings/java/android/app/src/main/res/mipmap-xhdpi/ic_launcher.png similarity index 100% rename from java/android/app/src/main/res/mipmap-xhdpi/ic_launcher.png rename to bindings/java/android/app/src/main/res/mipmap-xhdpi/ic_launcher.png diff --git a/java/android/app/src/main/res/mipmap-xhdpi/ic_launcher_round.png b/bindings/java/android/app/src/main/res/mipmap-xhdpi/ic_launcher_round.png similarity index 100% rename from java/android/app/src/main/res/mipmap-xhdpi/ic_launcher_round.png rename to bindings/java/android/app/src/main/res/mipmap-xhdpi/ic_launcher_round.png diff --git a/java/android/app/src/main/res/mipmap-xxhdpi/ic_launcher.png b/bindings/java/android/app/src/main/res/mipmap-xxhdpi/ic_launcher.png similarity index 100% rename from java/android/app/src/main/res/mipmap-xxhdpi/ic_launcher.png rename to bindings/java/android/app/src/main/res/mipmap-xxhdpi/ic_launcher.png diff --git a/java/android/app/src/main/res/mipmap-xxhdpi/ic_launcher_round.png b/bindings/java/android/app/src/main/res/mipmap-xxhdpi/ic_launcher_round.png similarity index 100% rename from java/android/app/src/main/res/mipmap-xxhdpi/ic_launcher_round.png rename to bindings/java/android/app/src/main/res/mipmap-xxhdpi/ic_launcher_round.png diff --git a/java/android/app/src/main/res/mipmap-xxxhdpi/ic_launcher.png b/bindings/java/android/app/src/main/res/mipmap-xxxhdpi/ic_launcher.png similarity index 100% rename from java/android/app/src/main/res/mipmap-xxxhdpi/ic_launcher.png rename to bindings/java/android/app/src/main/res/mipmap-xxxhdpi/ic_launcher.png diff --git a/java/android/app/src/main/res/mipmap-xxxhdpi/ic_launcher_round.png b/bindings/java/android/app/src/main/res/mipmap-xxxhdpi/ic_launcher_round.png similarity index 100% rename from java/android/app/src/main/res/mipmap-xxxhdpi/ic_launcher_round.png rename to bindings/java/android/app/src/main/res/mipmap-xxxhdpi/ic_launcher_round.png diff --git a/java/android/app/src/main/res/values/arrays.xml b/bindings/java/android/app/src/main/res/values/arrays.xml similarity index 100% rename from java/android/app/src/main/res/values/arrays.xml rename to bindings/java/android/app/src/main/res/values/arrays.xml diff --git a/java/android/app/src/main/res/values/colors.xml b/bindings/java/android/app/src/main/res/values/colors.xml similarity index 100% rename from java/android/app/src/main/res/values/colors.xml rename to bindings/java/android/app/src/main/res/values/colors.xml diff --git a/java/android/app/src/main/res/values/dimens.xml b/bindings/java/android/app/src/main/res/values/dimens.xml similarity index 100% rename from java/android/app/src/main/res/values/dimens.xml rename to bindings/java/android/app/src/main/res/values/dimens.xml diff --git a/java/android/app/src/main/res/values/strings.xml b/bindings/java/android/app/src/main/res/values/strings.xml similarity index 100% rename from java/android/app/src/main/res/values/strings.xml rename to bindings/java/android/app/src/main/res/values/strings.xml diff --git a/java/android/app/src/main/res/values/styles.xml b/bindings/java/android/app/src/main/res/values/styles.xml similarity index 100% rename from java/android/app/src/main/res/values/styles.xml rename to bindings/java/android/app/src/main/res/values/styles.xml diff --git a/java/android/app/src/main/res/values/values.xml b/bindings/java/android/app/src/main/res/values/values.xml similarity index 100% rename from java/android/app/src/main/res/values/values.xml rename to bindings/java/android/app/src/main/res/values/values.xml diff --git a/java/android/app/src/main/res/xml/classification_settings.xml b/bindings/java/android/app/src/main/res/xml/classification_settings.xml similarity index 100% rename from java/android/app/src/main/res/xml/classification_settings.xml rename to bindings/java/android/app/src/main/res/xml/classification_settings.xml diff --git a/java/android/app/src/main/res/xml/detection_settings.xml b/bindings/java/android/app/src/main/res/xml/detection_settings.xml similarity index 100% rename from java/android/app/src/main/res/xml/detection_settings.xml rename to bindings/java/android/app/src/main/res/xml/detection_settings.xml diff --git a/java/android/app/src/main/res/xml/face_align_settings.xml b/bindings/java/android/app/src/main/res/xml/face_align_settings.xml similarity index 100% rename from java/android/app/src/main/res/xml/face_align_settings.xml rename to bindings/java/android/app/src/main/res/xml/face_align_settings.xml diff --git a/java/android/app/src/main/res/xml/facedet_setting.xml b/bindings/java/android/app/src/main/res/xml/facedet_setting.xml similarity index 100% rename from java/android/app/src/main/res/xml/facedet_setting.xml rename to bindings/java/android/app/src/main/res/xml/facedet_setting.xml diff --git a/java/android/app/src/main/res/xml/keypointdetection_settting.xml b/bindings/java/android/app/src/main/res/xml/keypointdetection_settting.xml similarity index 100% rename from java/android/app/src/main/res/xml/keypointdetection_settting.xml rename to bindings/java/android/app/src/main/res/xml/keypointdetection_settting.xml diff --git a/java/android/app/src/main/res/xml/matting_settings.xml b/bindings/java/android/app/src/main/res/xml/matting_settings.xml similarity index 100% rename from java/android/app/src/main/res/xml/matting_settings.xml rename to bindings/java/android/app/src/main/res/xml/matting_settings.xml diff --git a/java/android/app/src/main/res/xml/ocr_settings.xml b/bindings/java/android/app/src/main/res/xml/ocr_settings.xml similarity index 100% rename from java/android/app/src/main/res/xml/ocr_settings.xml rename to bindings/java/android/app/src/main/res/xml/ocr_settings.xml diff --git a/java/android/app/src/main/res/xml/segmentation_setting.xml b/bindings/java/android/app/src/main/res/xml/segmentation_setting.xml similarity index 100% rename from java/android/app/src/main/res/xml/segmentation_setting.xml rename to bindings/java/android/app/src/main/res/xml/segmentation_setting.xml diff --git a/java/android/app/src/main/res/xml/super_resolution_setting.xml b/bindings/java/android/app/src/main/res/xml/super_resolution_setting.xml similarity index 100% rename from java/android/app/src/main/res/xml/super_resolution_setting.xml rename to bindings/java/android/app/src/main/res/xml/super_resolution_setting.xml diff --git a/java/android/app/src/main/res/xml/uie_settings.xml b/bindings/java/android/app/src/main/res/xml/uie_settings.xml similarity index 100% rename from java/android/app/src/main/res/xml/uie_settings.xml rename to bindings/java/android/app/src/main/res/xml/uie_settings.xml diff --git a/java/android/app/src/main/res/xml/voice_assistant_setting.xml b/bindings/java/android/app/src/main/res/xml/voice_assistant_setting.xml similarity index 100% rename from java/android/app/src/main/res/xml/voice_assistant_setting.xml rename to bindings/java/android/app/src/main/res/xml/voice_assistant_setting.xml diff --git a/java/android/build.gradle b/bindings/java/android/build.gradle similarity index 100% rename from java/android/build.gradle rename to bindings/java/android/build.gradle diff --git a/java/android/fastdeploy/.gitignore b/bindings/java/android/fastdeploy/.gitignore similarity index 100% rename from java/android/fastdeploy/.gitignore rename to bindings/java/android/fastdeploy/.gitignore diff --git a/java/android/fastdeploy/build.gradle b/bindings/java/android/fastdeploy/build.gradle similarity index 100% rename from java/android/fastdeploy/build.gradle rename to bindings/java/android/fastdeploy/build.gradle diff --git a/java/android/fastdeploy/consumer-rules.pro b/bindings/java/android/fastdeploy/consumer-rules.pro similarity index 100% rename from java/android/fastdeploy/consumer-rules.pro rename to bindings/java/android/fastdeploy/consumer-rules.pro diff --git a/java/android/fastdeploy/libs/.gitignore b/bindings/java/android/fastdeploy/libs/.gitignore similarity index 100% rename from java/android/fastdeploy/libs/.gitignore rename to bindings/java/android/fastdeploy/libs/.gitignore diff --git a/java/android/fastdeploy/proguard-rules.pro b/bindings/java/android/fastdeploy/proguard-rules.pro similarity index 100% rename from java/android/fastdeploy/proguard-rules.pro rename to bindings/java/android/fastdeploy/proguard-rules.pro diff --git a/java/android/fastdeploy/src/androidTest/java/com/baidu/paddle/fastdeploy/ExampleInstrumentedTest.java b/bindings/java/android/fastdeploy/src/androidTest/java/com/baidu/paddle/fastdeploy/ExampleInstrumentedTest.java similarity index 100% rename from java/android/fastdeploy/src/androidTest/java/com/baidu/paddle/fastdeploy/ExampleInstrumentedTest.java rename to bindings/java/android/fastdeploy/src/androidTest/java/com/baidu/paddle/fastdeploy/ExampleInstrumentedTest.java diff --git a/java/android/fastdeploy/src/main/AndroidManifest.xml b/bindings/java/android/fastdeploy/src/main/AndroidManifest.xml similarity index 100% rename from java/android/fastdeploy/src/main/AndroidManifest.xml rename to bindings/java/android/fastdeploy/src/main/AndroidManifest.xml diff --git a/java/android/fastdeploy/src/main/cpp/CMakeLists.txt b/bindings/java/android/fastdeploy/src/main/cpp/CMakeLists.txt similarity index 100% rename from java/android/fastdeploy/src/main/cpp/CMakeLists.txt rename to bindings/java/android/fastdeploy/src/main/cpp/CMakeLists.txt diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/assets_loader_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/assets_loader_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/assets_loader_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/assets_loader_jni.cc diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/assets_loader_jni.h b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/assets_loader_jni.h similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/assets_loader_jni.h rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/assets_loader_jni.h diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/bitmap_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/bitmap_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/bitmap_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/bitmap_jni.cc diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/bitmap_jni.h b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/bitmap_jni.h similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/bitmap_jni.h rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/bitmap_jni.h diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/convert_jni.h b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/convert_jni.h similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/convert_jni.h rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/convert_jni.h diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/perf_jni.h b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/perf_jni.h similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/perf_jni.h rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/perf_jni.h diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/pipeline/pipeline_utils_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/pipeline/pipeline_utils_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/pipeline/pipeline_utils_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/pipeline/pipeline_utils_jni.cc diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/pipeline/pipeline_utils_jni.h b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/pipeline/pipeline_utils_jni.h similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/pipeline/pipeline_utils_jni.h rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/pipeline/pipeline_utils_jni.h diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/pipeline/ppocr_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/pipeline/ppocr_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/pipeline/ppocr_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/pipeline/ppocr_jni.cc diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/runtime_option_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/runtime_option_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/runtime_option_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/runtime_option_jni.cc diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/runtime_option_jni.h b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/runtime_option_jni.h similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/runtime_option_jni.h rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/runtime_option_jni.h diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/text/text_results_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/text/text_results_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/text/text_results_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/text/text_results_jni.cc diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/text/text_results_jni.h b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/text/text_results_jni.h similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/text/text_results_jni.h rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/text/text_results_jni.h diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/text/uie/uie_model_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/text/uie/uie_model_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/text/uie/uie_model_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/text/uie/uie_model_jni.cc diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/text/uie/uie_utils_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/text/uie/uie_utils_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/text/uie/uie_utils_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/text/uie/uie_utils_jni.cc diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/text/uie/uie_utils_jni.h b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/text/uie/uie_utils_jni.h similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/text/uie/uie_utils_jni.h rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/text/uie/uie_utils_jni.h diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/classification/classification_utils_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/classification/classification_utils_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/classification/classification_utils_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/classification/classification_utils_jni.cc diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/classification/classification_utils_jni.h b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/classification/classification_utils_jni.h similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/classification/classification_utils_jni.h rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/classification/classification_utils_jni.h diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/classification/paddleclas_model_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/classification/paddleclas_model_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/classification/paddleclas_model_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/classification/paddleclas_model_jni.cc diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/detection/detection_utils_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/detection/detection_utils_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/detection/detection_utils_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/detection/detection_utils_jni.cc diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/detection/detection_utils_jni.h b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/detection/detection_utils_jni.h similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/detection/detection_utils_jni.h rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/detection/detection_utils_jni.h diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/detection/picodet_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/detection/picodet_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/detection/picodet_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/detection/picodet_jni.cc diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/facedet/facedet_utils_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/facedet/facedet_utils_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/facedet/facedet_utils_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/facedet/facedet_utils_jni.cc diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/facedet/facedet_utils_jni.h b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/facedet/facedet_utils_jni.h similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/facedet/facedet_utils_jni.h rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/facedet/facedet_utils_jni.h diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/facedet/scrfd_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/facedet/scrfd_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/facedet/scrfd_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/facedet/scrfd_jni.cc diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/facedet/yolov5face_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/facedet/yolov5face_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/facedet/yolov5face_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/facedet/yolov5face_jni.cc diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/keypointdetection/keypointdetection_utils_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/keypointdetection/keypointdetection_utils_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/keypointdetection/keypointdetection_utils_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/keypointdetection/keypointdetection_utils_jni.cc diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/keypointdetection/keypointdetection_utils_jni.h b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/keypointdetection/keypointdetection_utils_jni.h similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/keypointdetection/keypointdetection_utils_jni.h rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/keypointdetection/keypointdetection_utils_jni.h diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/keypointdetection/pptinypose_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/keypointdetection/pptinypose_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/keypointdetection/pptinypose_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/keypointdetection/pptinypose_jni.cc diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/results_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/results_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/results_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/results_jni.cc diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/results_jni.h b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/results_jni.h similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/results_jni.h rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/results_jni.h diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/segmentation/paddleseg_model_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/segmentation/paddleseg_model_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/segmentation/paddleseg_model_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/segmentation/paddleseg_model_jni.cc diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/segmentation/segmentation_utils_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/segmentation/segmentation_utils_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/segmentation/segmentation_utils_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/segmentation/segmentation_utils_jni.cc diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/segmentation/segmentation_utils_jni.h b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/segmentation/segmentation_utils_jni.h similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/segmentation/segmentation_utils_jni.h rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/segmentation/segmentation_utils_jni.h diff --git a/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/visualize_jni.cc b/bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/visualize_jni.cc similarity index 100% rename from java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/visualize_jni.cc rename to bindings/java/android/fastdeploy/src/main/cpp/fastdeploy_jni/vision/visualize_jni.cc diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/FDModelTag.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/FDModelTag.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/FDModelTag.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/FDModelTag.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/FastDeployInitializer.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/FastDeployInitializer.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/FastDeployInitializer.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/FastDeployInitializer.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/LitePowerMode.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/LitePowerMode.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/LitePowerMode.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/LitePowerMode.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/RuntimeOption.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/RuntimeOption.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/RuntimeOption.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/RuntimeOption.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/pipeline/PPOCRBase.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/pipeline/PPOCRBase.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/pipeline/PPOCRBase.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/pipeline/PPOCRBase.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/pipeline/PPOCRVersion.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/pipeline/PPOCRVersion.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/pipeline/PPOCRVersion.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/pipeline/PPOCRVersion.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/pipeline/PPOCRv2.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/pipeline/PPOCRv2.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/pipeline/PPOCRv2.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/pipeline/PPOCRv2.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/pipeline/PPOCRv3.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/pipeline/PPOCRv3.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/pipeline/PPOCRv3.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/pipeline/PPOCRv3.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/text/UIEResult.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/text/UIEResult.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/text/UIEResult.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/text/UIEResult.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/text/uie/SchemaLanguage.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/text/uie/SchemaLanguage.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/text/uie/SchemaLanguage.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/text/uie/SchemaLanguage.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/text/uie/SchemaNode.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/text/uie/SchemaNode.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/text/uie/SchemaNode.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/text/uie/SchemaNode.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/text/uie/UIEModel.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/text/uie/UIEModel.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/text/uie/UIEModel.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/text/uie/UIEModel.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/ClassifyResult.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/ClassifyResult.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/ClassifyResult.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/ClassifyResult.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/DetectionResult.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/DetectionResult.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/DetectionResult.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/DetectionResult.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/FaceDetectionResult.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/FaceDetectionResult.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/FaceDetectionResult.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/FaceDetectionResult.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/KeyPointDetectionResult.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/KeyPointDetectionResult.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/KeyPointDetectionResult.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/KeyPointDetectionResult.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/OCRResult.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/OCRResult.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/OCRResult.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/OCRResult.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/SegmentationResult.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/SegmentationResult.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/SegmentationResult.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/SegmentationResult.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/Visualize.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/Visualize.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/Visualize.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/Visualize.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/classification/PaddleClasModel.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/classification/PaddleClasModel.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/classification/PaddleClasModel.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/classification/PaddleClasModel.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/detection/PicoDet.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/detection/PicoDet.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/detection/PicoDet.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/detection/PicoDet.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/facedet/SCRFD.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/facedet/SCRFD.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/facedet/SCRFD.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/facedet/SCRFD.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/facedet/YOLOv5Face.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/facedet/YOLOv5Face.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/facedet/YOLOv5Face.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/facedet/YOLOv5Face.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/keypointdetection/PPTinyPose.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/keypointdetection/PPTinyPose.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/keypointdetection/PPTinyPose.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/keypointdetection/PPTinyPose.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/ocr/Classifier.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/ocr/Classifier.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/ocr/Classifier.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/ocr/Classifier.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/ocr/DBDetector.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/ocr/DBDetector.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/ocr/DBDetector.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/ocr/DBDetector.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/ocr/Recognizer.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/ocr/Recognizer.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/ocr/Recognizer.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/ocr/Recognizer.java diff --git a/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/segmentation/PaddleSegModel.java b/bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/segmentation/PaddleSegModel.java similarity index 100% rename from java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/segmentation/PaddleSegModel.java rename to bindings/java/android/fastdeploy/src/main/java/com/baidu/paddle/fastdeploy/vision/segmentation/PaddleSegModel.java diff --git a/java/android/fastdeploy/src/test/java/com/baidu/paddle/fastdeploy/ExampleUnitTest.java b/bindings/java/android/fastdeploy/src/test/java/com/baidu/paddle/fastdeploy/ExampleUnitTest.java similarity index 100% rename from java/android/fastdeploy/src/test/java/com/baidu/paddle/fastdeploy/ExampleUnitTest.java rename to bindings/java/android/fastdeploy/src/test/java/com/baidu/paddle/fastdeploy/ExampleUnitTest.java diff --git a/java/android/gradle.properties b/bindings/java/android/gradle.properties similarity index 100% rename from java/android/gradle.properties rename to bindings/java/android/gradle.properties diff --git a/java/android/gradle/wrapper/gradle-wrapper.jar b/bindings/java/android/gradle/wrapper/gradle-wrapper.jar similarity index 100% rename from java/android/gradle/wrapper/gradle-wrapper.jar rename to bindings/java/android/gradle/wrapper/gradle-wrapper.jar diff --git a/java/android/gradle/wrapper/gradle-wrapper.properties b/bindings/java/android/gradle/wrapper/gradle-wrapper.properties similarity index 100% rename from java/android/gradle/wrapper/gradle-wrapper.properties rename to bindings/java/android/gradle/wrapper/gradle-wrapper.properties diff --git a/java/android/gradlew b/bindings/java/android/gradlew similarity index 100% rename from java/android/gradlew rename to bindings/java/android/gradlew diff --git a/java/android/gradlew.bat b/bindings/java/android/gradlew.bat similarity index 100% rename from java/android/gradlew.bat rename to bindings/java/android/gradlew.bat diff --git a/java/android/local.properties b/bindings/java/android/local.properties similarity index 100% rename from java/android/local.properties rename to bindings/java/android/local.properties diff --git a/java/android/settings.gradle b/bindings/java/android/settings.gradle similarity index 100% rename from java/android/settings.gradle rename to bindings/java/android/settings.gradle diff --git a/java/android/ui/.gitignore b/bindings/java/android/ui/.gitignore similarity index 100% rename from java/android/ui/.gitignore rename to bindings/java/android/ui/.gitignore diff --git a/java/android/ui/build.gradle b/bindings/java/android/ui/build.gradle similarity index 100% rename from java/android/ui/build.gradle rename to bindings/java/android/ui/build.gradle diff --git a/java/android/ui/consumer-rules.pro b/bindings/java/android/ui/consumer-rules.pro similarity index 100% rename from java/android/ui/consumer-rules.pro rename to bindings/java/android/ui/consumer-rules.pro diff --git a/java/android/ui/local.properties b/bindings/java/android/ui/local.properties similarity index 100% rename from java/android/ui/local.properties rename to bindings/java/android/ui/local.properties diff --git a/java/android/ui/proguard-rules.pro b/bindings/java/android/ui/proguard-rules.pro similarity index 100% rename from java/android/ui/proguard-rules.pro rename to bindings/java/android/ui/proguard-rules.pro diff --git a/java/android/ui/src/main/AndroidManifest.xml b/bindings/java/android/ui/src/main/AndroidManifest.xml similarity index 100% rename from java/android/ui/src/main/AndroidManifest.xml rename to bindings/java/android/ui/src/main/AndroidManifest.xml diff --git a/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/Utils.java b/bindings/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/Utils.java similarity index 100% rename from java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/Utils.java rename to bindings/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/Utils.java diff --git a/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/layout/ActionBarLayout.java b/bindings/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/layout/ActionBarLayout.java similarity index 100% rename from java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/layout/ActionBarLayout.java rename to bindings/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/layout/ActionBarLayout.java diff --git a/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/view/AppCompatPreferenceActivity.java b/bindings/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/view/AppCompatPreferenceActivity.java similarity index 100% rename from java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/view/AppCompatPreferenceActivity.java rename to bindings/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/view/AppCompatPreferenceActivity.java diff --git a/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/view/CameraSurfaceView.java b/bindings/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/view/CameraSurfaceView.java similarity index 100% rename from java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/view/CameraSurfaceView.java rename to bindings/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/view/CameraSurfaceView.java diff --git a/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/view/ResultListView.java b/bindings/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/view/ResultListView.java similarity index 100% rename from java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/view/ResultListView.java rename to bindings/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/view/ResultListView.java diff --git a/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/view/adapter/BaseResultAdapter.java b/bindings/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/view/adapter/BaseResultAdapter.java similarity index 100% rename from java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/view/adapter/BaseResultAdapter.java rename to bindings/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/view/adapter/BaseResultAdapter.java diff --git a/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/view/model/BaseResultModel.java b/bindings/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/view/model/BaseResultModel.java similarity index 100% rename from java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/view/model/BaseResultModel.java rename to bindings/java/android/ui/src/main/java/com/baidu/paddle/fastdeploy/ui/view/model/BaseResultModel.java diff --git a/java/android/ui/src/main/res/drawable-v24/result_page_border_section_bk.xml b/bindings/java/android/ui/src/main/res/drawable-v24/result_page_border_section_bk.xml similarity index 100% rename from java/android/ui/src/main/res/drawable-v24/result_page_border_section_bk.xml rename to bindings/java/android/ui/src/main/res/drawable-v24/result_page_border_section_bk.xml diff --git a/java/android/ui/src/main/res/layout/base_result_page_item.xml b/bindings/java/android/ui/src/main/res/layout/base_result_page_item.xml similarity index 100% rename from java/android/ui/src/main/res/layout/base_result_page_item.xml rename to bindings/java/android/ui/src/main/res/layout/base_result_page_item.xml diff --git a/java/android/ui/src/main/res/values/colors.xml b/bindings/java/android/ui/src/main/res/values/colors.xml similarity index 100% rename from java/android/ui/src/main/res/values/colors.xml rename to bindings/java/android/ui/src/main/res/values/colors.xml diff --git a/java/android/ui/src/main/res/values/styles.xml b/bindings/java/android/ui/src/main/res/values/styles.xml similarity index 100% rename from java/android/ui/src/main/res/values/styles.xml rename to bindings/java/android/ui/src/main/res/values/styles.xml diff --git a/python/__init__.py b/bindings/python/__init__.py similarity index 100% rename from python/__init__.py rename to bindings/python/__init__.py diff --git a/python/fastdeploy/__init__.py b/bindings/python/fastdeploy/__init__.py similarity index 100% rename from python/fastdeploy/__init__.py rename to bindings/python/fastdeploy/__init__.py diff --git a/python/fastdeploy/c_lib_wrap.py.in b/bindings/python/fastdeploy/c_lib_wrap.py.in similarity index 100% rename from python/fastdeploy/c_lib_wrap.py.in rename to bindings/python/fastdeploy/c_lib_wrap.py.in diff --git a/python/fastdeploy/download.py b/bindings/python/fastdeploy/download.py similarity index 100% rename from python/fastdeploy/download.py rename to bindings/python/fastdeploy/download.py diff --git a/python/fastdeploy/encryption/__init__.py b/bindings/python/fastdeploy/encryption/__init__.py similarity index 100% rename from python/fastdeploy/encryption/__init__.py rename to bindings/python/fastdeploy/encryption/__init__.py diff --git a/python/fastdeploy/encryption/encryption.py b/bindings/python/fastdeploy/encryption/encryption.py similarity index 100% rename from python/fastdeploy/encryption/encryption.py rename to bindings/python/fastdeploy/encryption/encryption.py diff --git a/python/fastdeploy/libs/__init__.py b/bindings/python/fastdeploy/libs/__init__.py similarity index 100% rename from python/fastdeploy/libs/__init__.py rename to bindings/python/fastdeploy/libs/__init__.py diff --git a/python/fastdeploy/model.py b/bindings/python/fastdeploy/model.py similarity index 100% rename from python/fastdeploy/model.py rename to bindings/python/fastdeploy/model.py diff --git a/python/fastdeploy/pipeline/__init__.py b/bindings/python/fastdeploy/pipeline/__init__.py similarity index 100% rename from python/fastdeploy/pipeline/__init__.py rename to bindings/python/fastdeploy/pipeline/__init__.py diff --git a/python/fastdeploy/pipeline/pptinypose/__init__.py b/bindings/python/fastdeploy/pipeline/pptinypose/__init__.py similarity index 100% rename from python/fastdeploy/pipeline/pptinypose/__init__.py rename to bindings/python/fastdeploy/pipeline/pptinypose/__init__.py diff --git a/python/fastdeploy/runtime.py b/bindings/python/fastdeploy/runtime.py similarity index 100% rename from python/fastdeploy/runtime.py rename to bindings/python/fastdeploy/runtime.py diff --git a/python/fastdeploy/serving/__init__.py b/bindings/python/fastdeploy/serving/__init__.py similarity index 100% rename from python/fastdeploy/serving/__init__.py rename to bindings/python/fastdeploy/serving/__init__.py diff --git a/python/fastdeploy/serving/handler/__init__.py b/bindings/python/fastdeploy/serving/handler/__init__.py similarity index 100% rename from python/fastdeploy/serving/handler/__init__.py rename to bindings/python/fastdeploy/serving/handler/__init__.py diff --git a/python/fastdeploy/serving/handler/base_handler.py b/bindings/python/fastdeploy/serving/handler/base_handler.py similarity index 100% rename from python/fastdeploy/serving/handler/base_handler.py rename to bindings/python/fastdeploy/serving/handler/base_handler.py diff --git a/python/fastdeploy/serving/handler/vision_model_handler.py b/bindings/python/fastdeploy/serving/handler/vision_model_handler.py similarity index 100% rename from python/fastdeploy/serving/handler/vision_model_handler.py rename to bindings/python/fastdeploy/serving/handler/vision_model_handler.py diff --git a/python/fastdeploy/serving/model_manager.py b/bindings/python/fastdeploy/serving/model_manager.py similarity index 100% rename from python/fastdeploy/serving/model_manager.py rename to bindings/python/fastdeploy/serving/model_manager.py diff --git a/python/fastdeploy/serving/router/__init__.py b/bindings/python/fastdeploy/serving/router/__init__.py similarity index 100% rename from python/fastdeploy/serving/router/__init__.py rename to bindings/python/fastdeploy/serving/router/__init__.py diff --git a/python/fastdeploy/serving/router/base_router.py b/bindings/python/fastdeploy/serving/router/base_router.py similarity index 100% rename from python/fastdeploy/serving/router/base_router.py rename to bindings/python/fastdeploy/serving/router/base_router.py diff --git a/python/fastdeploy/serving/router/http_router.py b/bindings/python/fastdeploy/serving/router/http_router.py similarity index 100% rename from python/fastdeploy/serving/router/http_router.py rename to bindings/python/fastdeploy/serving/router/http_router.py diff --git a/python/fastdeploy/serving/server.py b/bindings/python/fastdeploy/serving/server.py similarity index 100% rename from python/fastdeploy/serving/server.py rename to bindings/python/fastdeploy/serving/server.py diff --git a/python/fastdeploy/serving/utils.py b/bindings/python/fastdeploy/serving/utils.py similarity index 100% rename from python/fastdeploy/serving/utils.py rename to bindings/python/fastdeploy/serving/utils.py diff --git a/python/fastdeploy/text/__init__.py b/bindings/python/fastdeploy/text/__init__.py similarity index 100% rename from python/fastdeploy/text/__init__.py rename to bindings/python/fastdeploy/text/__init__.py diff --git a/python/fastdeploy/text/uie/__init__.py b/bindings/python/fastdeploy/text/uie/__init__.py similarity index 100% rename from python/fastdeploy/text/uie/__init__.py rename to bindings/python/fastdeploy/text/uie/__init__.py diff --git a/python/fastdeploy/utils/__init__.py b/bindings/python/fastdeploy/utils/__init__.py similarity index 100% rename from python/fastdeploy/utils/__init__.py rename to bindings/python/fastdeploy/utils/__init__.py diff --git a/python/fastdeploy/utils/example_resource.py b/bindings/python/fastdeploy/utils/example_resource.py similarity index 100% rename from python/fastdeploy/utils/example_resource.py rename to bindings/python/fastdeploy/utils/example_resource.py diff --git a/python/fastdeploy/utils/hub_config.py b/bindings/python/fastdeploy/utils/hub_config.py similarity index 100% rename from python/fastdeploy/utils/hub_config.py rename to bindings/python/fastdeploy/utils/hub_config.py diff --git a/python/fastdeploy/utils/hub_env.py b/bindings/python/fastdeploy/utils/hub_env.py similarity index 100% rename from python/fastdeploy/utils/hub_env.py rename to bindings/python/fastdeploy/utils/hub_env.py diff --git a/python/fastdeploy/utils/hub_model_server.py b/bindings/python/fastdeploy/utils/hub_model_server.py similarity index 100% rename from python/fastdeploy/utils/hub_model_server.py rename to bindings/python/fastdeploy/utils/hub_model_server.py diff --git a/python/fastdeploy/vision/__init__.py b/bindings/python/fastdeploy/vision/__init__.py similarity index 100% rename from python/fastdeploy/vision/__init__.py rename to bindings/python/fastdeploy/vision/__init__.py diff --git a/python/fastdeploy/vision/classification/__init__.py b/bindings/python/fastdeploy/vision/classification/__init__.py similarity index 100% rename from python/fastdeploy/vision/classification/__init__.py rename to bindings/python/fastdeploy/vision/classification/__init__.py diff --git a/python/fastdeploy/vision/classification/contrib/__init__.py b/bindings/python/fastdeploy/vision/classification/contrib/__init__.py similarity index 100% rename from python/fastdeploy/vision/classification/contrib/__init__.py rename to bindings/python/fastdeploy/vision/classification/contrib/__init__.py diff --git a/python/fastdeploy/vision/classification/contrib/resnet.py b/bindings/python/fastdeploy/vision/classification/contrib/resnet.py similarity index 100% rename from python/fastdeploy/vision/classification/contrib/resnet.py rename to bindings/python/fastdeploy/vision/classification/contrib/resnet.py diff --git a/python/fastdeploy/vision/classification/contrib/yolov5cls.py b/bindings/python/fastdeploy/vision/classification/contrib/yolov5cls.py similarity index 100% rename from python/fastdeploy/vision/classification/contrib/yolov5cls.py rename to bindings/python/fastdeploy/vision/classification/contrib/yolov5cls.py diff --git a/python/fastdeploy/vision/classification/ppcls/__init__.py b/bindings/python/fastdeploy/vision/classification/ppcls/__init__.py similarity index 100% rename from python/fastdeploy/vision/classification/ppcls/__init__.py rename to bindings/python/fastdeploy/vision/classification/ppcls/__init__.py diff --git a/python/fastdeploy/vision/classification/ppshitu/__init__.py b/bindings/python/fastdeploy/vision/classification/ppshitu/__init__.py similarity index 100% rename from python/fastdeploy/vision/classification/ppshitu/__init__.py rename to bindings/python/fastdeploy/vision/classification/ppshitu/__init__.py diff --git a/python/fastdeploy/vision/common/__init__.py b/bindings/python/fastdeploy/vision/common/__init__.py similarity index 100% rename from python/fastdeploy/vision/common/__init__.py rename to bindings/python/fastdeploy/vision/common/__init__.py diff --git a/python/fastdeploy/vision/common/manager.py b/bindings/python/fastdeploy/vision/common/manager.py similarity index 100% rename from python/fastdeploy/vision/common/manager.py rename to bindings/python/fastdeploy/vision/common/manager.py diff --git a/python/fastdeploy/vision/common/processors.py b/bindings/python/fastdeploy/vision/common/processors.py similarity index 100% rename from python/fastdeploy/vision/common/processors.py rename to bindings/python/fastdeploy/vision/common/processors.py diff --git a/python/fastdeploy/vision/detection/__init__.py b/bindings/python/fastdeploy/vision/detection/__init__.py similarity index 100% rename from python/fastdeploy/vision/detection/__init__.py rename to bindings/python/fastdeploy/vision/detection/__init__.py diff --git a/python/fastdeploy/vision/detection/contrib/__init__.py b/bindings/python/fastdeploy/vision/detection/contrib/__init__.py similarity index 100% rename from python/fastdeploy/vision/detection/contrib/__init__.py rename to bindings/python/fastdeploy/vision/detection/contrib/__init__.py diff --git a/python/fastdeploy/vision/detection/contrib/fastestdet.py b/bindings/python/fastdeploy/vision/detection/contrib/fastestdet.py similarity index 100% rename from python/fastdeploy/vision/detection/contrib/fastestdet.py rename to bindings/python/fastdeploy/vision/detection/contrib/fastestdet.py diff --git a/python/fastdeploy/vision/detection/contrib/nanodet_plus.py b/bindings/python/fastdeploy/vision/detection/contrib/nanodet_plus.py similarity index 100% rename from python/fastdeploy/vision/detection/contrib/nanodet_plus.py rename to bindings/python/fastdeploy/vision/detection/contrib/nanodet_plus.py diff --git a/python/fastdeploy/vision/detection/contrib/rkyolo/__init__.py b/bindings/python/fastdeploy/vision/detection/contrib/rkyolo/__init__.py similarity index 100% rename from python/fastdeploy/vision/detection/contrib/rkyolo/__init__.py rename to bindings/python/fastdeploy/vision/detection/contrib/rkyolo/__init__.py diff --git a/python/fastdeploy/vision/detection/contrib/rkyolo/rkyolov5.py b/bindings/python/fastdeploy/vision/detection/contrib/rkyolo/rkyolov5.py similarity index 100% rename from python/fastdeploy/vision/detection/contrib/rkyolo/rkyolov5.py rename to bindings/python/fastdeploy/vision/detection/contrib/rkyolo/rkyolov5.py diff --git a/python/fastdeploy/vision/detection/contrib/scaled_yolov4.py b/bindings/python/fastdeploy/vision/detection/contrib/scaled_yolov4.py similarity index 100% rename from python/fastdeploy/vision/detection/contrib/scaled_yolov4.py rename to bindings/python/fastdeploy/vision/detection/contrib/scaled_yolov4.py diff --git a/python/fastdeploy/vision/detection/contrib/yolor.py b/bindings/python/fastdeploy/vision/detection/contrib/yolor.py similarity index 100% rename from python/fastdeploy/vision/detection/contrib/yolor.py rename to bindings/python/fastdeploy/vision/detection/contrib/yolor.py diff --git a/python/fastdeploy/vision/detection/contrib/yolov5.py b/bindings/python/fastdeploy/vision/detection/contrib/yolov5.py similarity index 100% rename from python/fastdeploy/vision/detection/contrib/yolov5.py rename to bindings/python/fastdeploy/vision/detection/contrib/yolov5.py diff --git a/python/fastdeploy/vision/detection/contrib/yolov5lite.py b/bindings/python/fastdeploy/vision/detection/contrib/yolov5lite.py similarity index 100% rename from python/fastdeploy/vision/detection/contrib/yolov5lite.py rename to bindings/python/fastdeploy/vision/detection/contrib/yolov5lite.py diff --git a/python/fastdeploy/vision/detection/contrib/yolov5seg.py b/bindings/python/fastdeploy/vision/detection/contrib/yolov5seg.py similarity index 100% rename from python/fastdeploy/vision/detection/contrib/yolov5seg.py rename to bindings/python/fastdeploy/vision/detection/contrib/yolov5seg.py diff --git a/python/fastdeploy/vision/detection/contrib/yolov6.py b/bindings/python/fastdeploy/vision/detection/contrib/yolov6.py similarity index 100% rename from python/fastdeploy/vision/detection/contrib/yolov6.py rename to bindings/python/fastdeploy/vision/detection/contrib/yolov6.py diff --git a/python/fastdeploy/vision/detection/contrib/yolov7.py b/bindings/python/fastdeploy/vision/detection/contrib/yolov7.py similarity index 100% rename from python/fastdeploy/vision/detection/contrib/yolov7.py rename to bindings/python/fastdeploy/vision/detection/contrib/yolov7.py diff --git a/python/fastdeploy/vision/detection/contrib/yolov7end2end_ort.py b/bindings/python/fastdeploy/vision/detection/contrib/yolov7end2end_ort.py similarity index 100% rename from python/fastdeploy/vision/detection/contrib/yolov7end2end_ort.py rename to bindings/python/fastdeploy/vision/detection/contrib/yolov7end2end_ort.py diff --git a/python/fastdeploy/vision/detection/contrib/yolov7end2end_trt.py b/bindings/python/fastdeploy/vision/detection/contrib/yolov7end2end_trt.py similarity index 100% rename from python/fastdeploy/vision/detection/contrib/yolov7end2end_trt.py rename to bindings/python/fastdeploy/vision/detection/contrib/yolov7end2end_trt.py diff --git a/python/fastdeploy/vision/detection/contrib/yolov8.py b/bindings/python/fastdeploy/vision/detection/contrib/yolov8.py similarity index 100% rename from python/fastdeploy/vision/detection/contrib/yolov8.py rename to bindings/python/fastdeploy/vision/detection/contrib/yolov8.py diff --git a/python/fastdeploy/vision/detection/contrib/yolox.py b/bindings/python/fastdeploy/vision/detection/contrib/yolox.py similarity index 100% rename from python/fastdeploy/vision/detection/contrib/yolox.py rename to bindings/python/fastdeploy/vision/detection/contrib/yolox.py diff --git a/python/fastdeploy/vision/detection/ppdet/__init__.py b/bindings/python/fastdeploy/vision/detection/ppdet/__init__.py similarity index 100% rename from python/fastdeploy/vision/detection/ppdet/__init__.py rename to bindings/python/fastdeploy/vision/detection/ppdet/__init__.py diff --git a/python/fastdeploy/vision/evaluation/__init__.py b/bindings/python/fastdeploy/vision/evaluation/__init__.py similarity index 100% rename from python/fastdeploy/vision/evaluation/__init__.py rename to bindings/python/fastdeploy/vision/evaluation/__init__.py diff --git a/python/fastdeploy/vision/evaluation/classify.py b/bindings/python/fastdeploy/vision/evaluation/classify.py similarity index 100% rename from python/fastdeploy/vision/evaluation/classify.py rename to bindings/python/fastdeploy/vision/evaluation/classify.py diff --git a/python/fastdeploy/vision/evaluation/detection.py b/bindings/python/fastdeploy/vision/evaluation/detection.py similarity index 100% rename from python/fastdeploy/vision/evaluation/detection.py rename to bindings/python/fastdeploy/vision/evaluation/detection.py diff --git a/python/fastdeploy/vision/evaluation/segmentation.py b/bindings/python/fastdeploy/vision/evaluation/segmentation.py similarity index 100% rename from python/fastdeploy/vision/evaluation/segmentation.py rename to bindings/python/fastdeploy/vision/evaluation/segmentation.py diff --git a/python/fastdeploy/vision/evaluation/utils/__init__.py b/bindings/python/fastdeploy/vision/evaluation/utils/__init__.py similarity index 100% rename from python/fastdeploy/vision/evaluation/utils/__init__.py rename to bindings/python/fastdeploy/vision/evaluation/utils/__init__.py diff --git a/python/fastdeploy/vision/evaluation/utils/cityscapes.py b/bindings/python/fastdeploy/vision/evaluation/utils/cityscapes.py similarity index 100% rename from python/fastdeploy/vision/evaluation/utils/cityscapes.py rename to bindings/python/fastdeploy/vision/evaluation/utils/cityscapes.py diff --git a/python/fastdeploy/vision/evaluation/utils/coco.py b/bindings/python/fastdeploy/vision/evaluation/utils/coco.py similarity index 100% rename from python/fastdeploy/vision/evaluation/utils/coco.py rename to bindings/python/fastdeploy/vision/evaluation/utils/coco.py diff --git a/python/fastdeploy/vision/evaluation/utils/coco_metrics.py b/bindings/python/fastdeploy/vision/evaluation/utils/coco_metrics.py similarity index 100% rename from python/fastdeploy/vision/evaluation/utils/coco_metrics.py rename to bindings/python/fastdeploy/vision/evaluation/utils/coco_metrics.py diff --git a/python/fastdeploy/vision/evaluation/utils/coco_utils.py b/bindings/python/fastdeploy/vision/evaluation/utils/coco_utils.py similarity index 100% rename from python/fastdeploy/vision/evaluation/utils/coco_utils.py rename to bindings/python/fastdeploy/vision/evaluation/utils/coco_utils.py diff --git a/python/fastdeploy/vision/evaluation/utils/fd_logging.py b/bindings/python/fastdeploy/vision/evaluation/utils/fd_logging.py similarity index 100% rename from python/fastdeploy/vision/evaluation/utils/fd_logging.py rename to bindings/python/fastdeploy/vision/evaluation/utils/fd_logging.py diff --git a/python/fastdeploy/vision/evaluation/utils/json_results.py b/bindings/python/fastdeploy/vision/evaluation/utils/json_results.py similarity index 100% rename from python/fastdeploy/vision/evaluation/utils/json_results.py rename to bindings/python/fastdeploy/vision/evaluation/utils/json_results.py diff --git a/python/fastdeploy/vision/evaluation/utils/map_utils.py b/bindings/python/fastdeploy/vision/evaluation/utils/map_utils.py similarity index 100% rename from python/fastdeploy/vision/evaluation/utils/map_utils.py rename to bindings/python/fastdeploy/vision/evaluation/utils/map_utils.py diff --git a/python/fastdeploy/vision/evaluation/utils/seg_metrics.py b/bindings/python/fastdeploy/vision/evaluation/utils/seg_metrics.py similarity index 100% rename from python/fastdeploy/vision/evaluation/utils/seg_metrics.py rename to bindings/python/fastdeploy/vision/evaluation/utils/seg_metrics.py diff --git a/python/fastdeploy/vision/evaluation/utils/util.py b/bindings/python/fastdeploy/vision/evaluation/utils/util.py similarity index 100% rename from python/fastdeploy/vision/evaluation/utils/util.py rename to bindings/python/fastdeploy/vision/evaluation/utils/util.py diff --git a/python/fastdeploy/vision/facealign/__init__.py b/bindings/python/fastdeploy/vision/facealign/__init__.py similarity index 100% rename from python/fastdeploy/vision/facealign/__init__.py rename to bindings/python/fastdeploy/vision/facealign/__init__.py diff --git a/python/fastdeploy/vision/facealign/contrib/__init__.py b/bindings/python/fastdeploy/vision/facealign/contrib/__init__.py similarity index 100% rename from python/fastdeploy/vision/facealign/contrib/__init__.py rename to bindings/python/fastdeploy/vision/facealign/contrib/__init__.py diff --git a/python/fastdeploy/vision/facealign/contrib/face_landmark_1000.py b/bindings/python/fastdeploy/vision/facealign/contrib/face_landmark_1000.py similarity index 100% rename from python/fastdeploy/vision/facealign/contrib/face_landmark_1000.py rename to bindings/python/fastdeploy/vision/facealign/contrib/face_landmark_1000.py diff --git a/python/fastdeploy/vision/facealign/contrib/pfld.py b/bindings/python/fastdeploy/vision/facealign/contrib/pfld.py similarity index 100% rename from python/fastdeploy/vision/facealign/contrib/pfld.py rename to bindings/python/fastdeploy/vision/facealign/contrib/pfld.py diff --git a/python/fastdeploy/vision/facealign/contrib/pipnet.py b/bindings/python/fastdeploy/vision/facealign/contrib/pipnet.py similarity index 100% rename from python/fastdeploy/vision/facealign/contrib/pipnet.py rename to bindings/python/fastdeploy/vision/facealign/contrib/pipnet.py diff --git a/python/fastdeploy/vision/facedet/__init__.py b/bindings/python/fastdeploy/vision/facedet/__init__.py similarity index 100% rename from python/fastdeploy/vision/facedet/__init__.py rename to bindings/python/fastdeploy/vision/facedet/__init__.py diff --git a/python/fastdeploy/vision/facedet/contrib/__init__.py b/bindings/python/fastdeploy/vision/facedet/contrib/__init__.py similarity index 100% rename from python/fastdeploy/vision/facedet/contrib/__init__.py rename to bindings/python/fastdeploy/vision/facedet/contrib/__init__.py diff --git a/python/fastdeploy/vision/facedet/contrib/blazeface.py b/bindings/python/fastdeploy/vision/facedet/contrib/blazeface.py similarity index 100% rename from python/fastdeploy/vision/facedet/contrib/blazeface.py rename to bindings/python/fastdeploy/vision/facedet/contrib/blazeface.py diff --git a/python/fastdeploy/vision/facedet/contrib/centerface.py b/bindings/python/fastdeploy/vision/facedet/contrib/centerface.py similarity index 100% rename from python/fastdeploy/vision/facedet/contrib/centerface.py rename to bindings/python/fastdeploy/vision/facedet/contrib/centerface.py diff --git a/python/fastdeploy/vision/facedet/contrib/retinaface.py b/bindings/python/fastdeploy/vision/facedet/contrib/retinaface.py similarity index 100% rename from python/fastdeploy/vision/facedet/contrib/retinaface.py rename to bindings/python/fastdeploy/vision/facedet/contrib/retinaface.py diff --git a/python/fastdeploy/vision/facedet/contrib/scrfd.py b/bindings/python/fastdeploy/vision/facedet/contrib/scrfd.py similarity index 100% rename from python/fastdeploy/vision/facedet/contrib/scrfd.py rename to bindings/python/fastdeploy/vision/facedet/contrib/scrfd.py diff --git a/python/fastdeploy/vision/facedet/contrib/ultraface.py b/bindings/python/fastdeploy/vision/facedet/contrib/ultraface.py similarity index 100% rename from python/fastdeploy/vision/facedet/contrib/ultraface.py rename to bindings/python/fastdeploy/vision/facedet/contrib/ultraface.py diff --git a/python/fastdeploy/vision/facedet/contrib/yolov5face.py b/bindings/python/fastdeploy/vision/facedet/contrib/yolov5face.py similarity index 100% rename from python/fastdeploy/vision/facedet/contrib/yolov5face.py rename to bindings/python/fastdeploy/vision/facedet/contrib/yolov5face.py diff --git a/python/fastdeploy/vision/facedet/contrib/yolov7face.py b/bindings/python/fastdeploy/vision/facedet/contrib/yolov7face.py similarity index 100% rename from python/fastdeploy/vision/facedet/contrib/yolov7face.py rename to bindings/python/fastdeploy/vision/facedet/contrib/yolov7face.py diff --git a/python/fastdeploy/vision/faceid/__init__.py b/bindings/python/fastdeploy/vision/faceid/__init__.py similarity index 100% rename from python/fastdeploy/vision/faceid/__init__.py rename to bindings/python/fastdeploy/vision/faceid/__init__.py diff --git a/python/fastdeploy/vision/faceid/contrib/__init__.py b/bindings/python/fastdeploy/vision/faceid/contrib/__init__.py similarity index 100% rename from python/fastdeploy/vision/faceid/contrib/__init__.py rename to bindings/python/fastdeploy/vision/faceid/contrib/__init__.py diff --git a/python/fastdeploy/vision/faceid/contrib/adaface/__init__.py b/bindings/python/fastdeploy/vision/faceid/contrib/adaface/__init__.py similarity index 100% rename from python/fastdeploy/vision/faceid/contrib/adaface/__init__.py rename to bindings/python/fastdeploy/vision/faceid/contrib/adaface/__init__.py diff --git a/python/fastdeploy/vision/faceid/contrib/insightface/__init__.py b/bindings/python/fastdeploy/vision/faceid/contrib/insightface/__init__.py similarity index 100% rename from python/fastdeploy/vision/faceid/contrib/insightface/__init__.py rename to bindings/python/fastdeploy/vision/faceid/contrib/insightface/__init__.py diff --git a/python/fastdeploy/vision/generation/__init__.py b/bindings/python/fastdeploy/vision/generation/__init__.py similarity index 100% rename from python/fastdeploy/vision/generation/__init__.py rename to bindings/python/fastdeploy/vision/generation/__init__.py diff --git a/python/fastdeploy/vision/generation/contrib/__init__.py b/bindings/python/fastdeploy/vision/generation/contrib/__init__.py similarity index 100% rename from python/fastdeploy/vision/generation/contrib/__init__.py rename to bindings/python/fastdeploy/vision/generation/contrib/__init__.py diff --git a/python/fastdeploy/vision/generation/contrib/anemigan.py b/bindings/python/fastdeploy/vision/generation/contrib/anemigan.py similarity index 100% rename from python/fastdeploy/vision/generation/contrib/anemigan.py rename to bindings/python/fastdeploy/vision/generation/contrib/anemigan.py diff --git a/python/fastdeploy/vision/headpose/__init__.py b/bindings/python/fastdeploy/vision/headpose/__init__.py similarity index 100% rename from python/fastdeploy/vision/headpose/__init__.py rename to bindings/python/fastdeploy/vision/headpose/__init__.py diff --git a/python/fastdeploy/vision/headpose/contrib/__init__.py b/bindings/python/fastdeploy/vision/headpose/contrib/__init__.py similarity index 100% rename from python/fastdeploy/vision/headpose/contrib/__init__.py rename to bindings/python/fastdeploy/vision/headpose/contrib/__init__.py diff --git a/python/fastdeploy/vision/headpose/contrib/fsanet.py b/bindings/python/fastdeploy/vision/headpose/contrib/fsanet.py similarity index 100% rename from python/fastdeploy/vision/headpose/contrib/fsanet.py rename to bindings/python/fastdeploy/vision/headpose/contrib/fsanet.py diff --git a/python/fastdeploy/vision/keypointdetection/__init__.py b/bindings/python/fastdeploy/vision/keypointdetection/__init__.py similarity index 100% rename from python/fastdeploy/vision/keypointdetection/__init__.py rename to bindings/python/fastdeploy/vision/keypointdetection/__init__.py diff --git a/python/fastdeploy/vision/keypointdetection/pptinypose/__init__.py b/bindings/python/fastdeploy/vision/keypointdetection/pptinypose/__init__.py similarity index 100% rename from python/fastdeploy/vision/keypointdetection/pptinypose/__init__.py rename to bindings/python/fastdeploy/vision/keypointdetection/pptinypose/__init__.py diff --git a/python/fastdeploy/vision/matting/__init__.py b/bindings/python/fastdeploy/vision/matting/__init__.py similarity index 100% rename from python/fastdeploy/vision/matting/__init__.py rename to bindings/python/fastdeploy/vision/matting/__init__.py diff --git a/python/fastdeploy/vision/matting/contrib/__init__.py b/bindings/python/fastdeploy/vision/matting/contrib/__init__.py similarity index 100% rename from python/fastdeploy/vision/matting/contrib/__init__.py rename to bindings/python/fastdeploy/vision/matting/contrib/__init__.py diff --git a/python/fastdeploy/vision/matting/contrib/modnet.py b/bindings/python/fastdeploy/vision/matting/contrib/modnet.py similarity index 100% rename from python/fastdeploy/vision/matting/contrib/modnet.py rename to bindings/python/fastdeploy/vision/matting/contrib/modnet.py diff --git a/python/fastdeploy/vision/matting/contrib/rvm.py b/bindings/python/fastdeploy/vision/matting/contrib/rvm.py similarity index 100% rename from python/fastdeploy/vision/matting/contrib/rvm.py rename to bindings/python/fastdeploy/vision/matting/contrib/rvm.py diff --git a/python/fastdeploy/vision/matting/ppmatting/__init__.py b/bindings/python/fastdeploy/vision/matting/ppmatting/__init__.py similarity index 100% rename from python/fastdeploy/vision/matting/ppmatting/__init__.py rename to bindings/python/fastdeploy/vision/matting/ppmatting/__init__.py diff --git a/python/fastdeploy/vision/ocr/__init__.py b/bindings/python/fastdeploy/vision/ocr/__init__.py similarity index 100% rename from python/fastdeploy/vision/ocr/__init__.py rename to bindings/python/fastdeploy/vision/ocr/__init__.py diff --git a/python/fastdeploy/vision/ocr/ppocr/__init__.py b/bindings/python/fastdeploy/vision/ocr/ppocr/__init__.py similarity index 88% rename from python/fastdeploy/vision/ocr/ppocr/__init__.py rename to bindings/python/fastdeploy/vision/ocr/ppocr/__init__.py index 272c93b22c8..c26562cee8d 100755 --- a/python/fastdeploy/vision/ocr/ppocr/__init__.py +++ b/bindings/python/fastdeploy/vision/ocr/ppocr/__init__.py @@ -191,6 +191,24 @@ def use_dilation(self, value): bool), "The value to set `use_dilation` must be type of bool." self._postprocessor.use_dilation = value + @property + def det_db_max_candidates(self): + """ + Return the det_db_max_candidates of DBDetectorPostprocessor + """ + return self._postprocessor.det_db_max_candidates + + @det_db_max_candidates.setter + def det_db_max_candidates(self, value): + """Set the det_db_max_candidates for DBDetectorPostprocessor + + :param: value : the det_db_max_candidates value + """ + assert isinstance( + value, + int), "The value to set `det_db_max_candidates` must be type of int." + self._postprocessor.det_db_max_candidates = value + class DBDetector(FastDeployModel): def __init__(self, @@ -854,6 +872,158 @@ def postprocessor(self, value): self._model.postprocessor = value +class PPOCRv6(FastDeployModel): + def __init__(self, det_model=None, cls_model=None, rec_model=None): + """Consruct a pipeline with text detector, direction classifier and text recognizer models + + :param det_model: (FastDeployModel) The detection model object created by fastdeploy.vision.ocr.DBDetector. + :param cls_model: (FastDeployModel) The classification model object created by fastdeploy.vision.ocr.Classifier. + :param rec_model: (FastDeployModel) The recognition model object created by fastdeploy.vision.ocr.Recognizer. + """ + assert det_model is not None and rec_model is not None, "The det_model and rec_model cannot be None." + if cls_model is None: + self.system_ = C.vision.ocr.PPOCRv6(det_model._model, + rec_model._model) + else: + self.system_ = C.vision.ocr.PPOCRv6( + det_model._model, cls_model._model, rec_model._model) + + def clone(self): + """Clone PPOCRv6 pipeline object + :return: a new PPOCRv6 pipeline object + """ + + class PPOCRv6Clone(PPOCRv6): + def __init__(self, system): + self.system_ = system + + clone_model = PPOCRv6Clone(self.system_.clone()) + return clone_model + + def predict(self, input_image): + """Predict an input image + :param input_image: (numpy.ndarray)The input image data, 3-D array with layout HWC, BGR format + :return: OCRResult + """ + return self.system_.predict(input_image) + + def batch_predict(self, images): + """Predict a batch of input image + :param images: (list of numpy.ndarray) The input image list, each element is a 3-D array with layout HWC, BGR format + :return: OCRBatchResult + """ + return self.system_.batch_predict(images) + + @property + def cls_batch_size(self): + return self.system_.cls_batch_size + + @cls_batch_size.setter + def cls_batch_size(self, value): + assert isinstance( + value, + int), "The value to set `cls_batch_size` must be type of int." + self.system_.cls_batch_size = value + + @property + def rec_batch_size(self): + return self.system_.rec_batch_size + + @rec_batch_size.setter + def rec_batch_size(self, value): + assert isinstance( + value, + int), "The value to set `rec_batch_size` must be type of int." + self.system_.rec_batch_size = value + + +class PPOCRSystemv6(PPOCRv6): + def __init__(self, det_model=None, cls_model=None, rec_model=None): + logging.warning( + "DEPRECATED: fd.vision.ocr.PPOCRSystemv6 is deprecated, " + "please use fd.vision.ocr.PPOCRv6 instead.") + super(PPOCRSystemv6, self).__init__(det_model, cls_model, rec_model) + + def predict(self, input_image): + return super(PPOCRSystemv6, self).predict(input_image) + + +class PPOCRv5(FastDeployModel): + def __init__(self, det_model=None, cls_model=None, rec_model=None): + """Consruct a pipeline with text detector, direction classifier and text recognizer models + + :param det_model: (FastDeployModel) The detection model object created by fastdeploy.vision.ocr.DBDetector. + :param cls_model: (FastDeployModel) The classification model object created by fastdeploy.vision.ocr.Classifier. + :param rec_model: (FastDeployModel) The recognition model object created by fastdeploy.vision.ocr.Recognizer. + """ + assert det_model is not None and rec_model is not None, "The det_model and rec_model cannot be None." + if cls_model is None: + self.system_ = C.vision.ocr.PPOCRv5(det_model._model, + rec_model._model) + else: + self.system_ = C.vision.ocr.PPOCRv5( + det_model._model, cls_model._model, rec_model._model) + + def clone(self): + """Clone PPOCRv5 pipeline object + :return: a new PPOCRv5 pipeline object + """ + + class PPOCRv5Clone(PPOCRv5): + def __init__(self, system): + self.system_ = system + + clone_model = PPOCRv5Clone(self.system_.clone()) + return clone_model + + def predict(self, input_image): + """Predict an input image + :param input_image: (numpy.ndarray)The input image data, 3-D array with layout HWC, BGR format + :return: OCRResult + """ + return self.system_.predict(input_image) + + def batch_predict(self, images): + """Predict a batch of input image + :param images: (list of numpy.ndarray) The input image list, each element is a 3-D array with layout HWC, BGR format + :return: OCRBatchResult + """ + return self.system_.batch_predict(images) + + @property + def cls_batch_size(self): + return self.system_.cls_batch_size + + @cls_batch_size.setter + def cls_batch_size(self, value): + assert isinstance( + value, + int), "The value to set `cls_batch_size` must be type of int." + self.system_.cls_batch_size = value + + @property + def rec_batch_size(self): + return self.system_.rec_batch_size + + @rec_batch_size.setter + def rec_batch_size(self, value): + assert isinstance( + value, + int), "The value to set `rec_batch_size` must be type of int." + self.system_.rec_batch_size = value + + +class PPOCRSystemv5(PPOCRv5): + def __init__(self, det_model=None, cls_model=None, rec_model=None): + logging.warning( + "DEPRECATED: fd.vision.ocr.PPOCRSystemv5 is deprecated, " + "please use fd.vision.ocr.PPOCRv5 instead.") + super(PPOCRSystemv5, self).__init__(det_model, cls_model, rec_model) + + def predict(self, input_image): + return super(PPOCRSystemv5, self).predict(input_image) + + class PPOCRv4(FastDeployModel): def __init__(self, det_model=None, cls_model=None, rec_model=None): """Consruct a pipeline with text detector, direction classifier and text recognizer models diff --git a/python/fastdeploy/vision/ocr/ppocr/utils/__init__.py b/bindings/python/fastdeploy/vision/ocr/ppocr/utils/__init__.py similarity index 100% rename from python/fastdeploy/vision/ocr/ppocr/utils/__init__.py rename to bindings/python/fastdeploy/vision/ocr/ppocr/utils/__init__.py diff --git a/python/fastdeploy/vision/ocr/ppocr/utils/ser_vi_layoutxlm/__init__.py b/bindings/python/fastdeploy/vision/ocr/ppocr/utils/ser_vi_layoutxlm/__init__.py similarity index 100% rename from python/fastdeploy/vision/ocr/ppocr/utils/ser_vi_layoutxlm/__init__.py rename to bindings/python/fastdeploy/vision/ocr/ppocr/utils/ser_vi_layoutxlm/__init__.py diff --git a/python/fastdeploy/vision/ocr/ppocr/utils/ser_vi_layoutxlm/operators.py b/bindings/python/fastdeploy/vision/ocr/ppocr/utils/ser_vi_layoutxlm/operators.py similarity index 100% rename from python/fastdeploy/vision/ocr/ppocr/utils/ser_vi_layoutxlm/operators.py rename to bindings/python/fastdeploy/vision/ocr/ppocr/utils/ser_vi_layoutxlm/operators.py diff --git a/python/fastdeploy/vision/ocr/ppocr/utils/ser_vi_layoutxlm/transforms.py b/bindings/python/fastdeploy/vision/ocr/ppocr/utils/ser_vi_layoutxlm/transforms.py similarity index 100% rename from python/fastdeploy/vision/ocr/ppocr/utils/ser_vi_layoutxlm/transforms.py rename to bindings/python/fastdeploy/vision/ocr/ppocr/utils/ser_vi_layoutxlm/transforms.py diff --git a/python/fastdeploy/vision/ocr/ppocr/utils/ser_vi_layoutxlm/vqa_utils.py b/bindings/python/fastdeploy/vision/ocr/ppocr/utils/ser_vi_layoutxlm/vqa_utils.py similarity index 100% rename from python/fastdeploy/vision/ocr/ppocr/utils/ser_vi_layoutxlm/vqa_utils.py rename to bindings/python/fastdeploy/vision/ocr/ppocr/utils/ser_vi_layoutxlm/vqa_utils.py diff --git a/python/fastdeploy/vision/perception/__init__.py b/bindings/python/fastdeploy/vision/perception/__init__.py similarity index 100% rename from python/fastdeploy/vision/perception/__init__.py rename to bindings/python/fastdeploy/vision/perception/__init__.py diff --git a/python/fastdeploy/vision/perception/paddle3d/__init__.py b/bindings/python/fastdeploy/vision/perception/paddle3d/__init__.py similarity index 100% rename from python/fastdeploy/vision/perception/paddle3d/__init__.py rename to bindings/python/fastdeploy/vision/perception/paddle3d/__init__.py diff --git a/python/fastdeploy/vision/perception/paddle3d/caddn.py b/bindings/python/fastdeploy/vision/perception/paddle3d/caddn.py similarity index 100% rename from python/fastdeploy/vision/perception/paddle3d/caddn.py rename to bindings/python/fastdeploy/vision/perception/paddle3d/caddn.py diff --git a/python/fastdeploy/vision/perception/paddle3d/centerpoint.py b/bindings/python/fastdeploy/vision/perception/paddle3d/centerpoint.py similarity index 100% rename from python/fastdeploy/vision/perception/paddle3d/centerpoint.py rename to bindings/python/fastdeploy/vision/perception/paddle3d/centerpoint.py diff --git a/python/fastdeploy/vision/perception/paddle3d/petr.py b/bindings/python/fastdeploy/vision/perception/paddle3d/petr.py similarity index 100% rename from python/fastdeploy/vision/perception/paddle3d/petr.py rename to bindings/python/fastdeploy/vision/perception/paddle3d/petr.py diff --git a/python/fastdeploy/vision/perception/paddle3d/smoke.py b/bindings/python/fastdeploy/vision/perception/paddle3d/smoke.py similarity index 100% rename from python/fastdeploy/vision/perception/paddle3d/smoke.py rename to bindings/python/fastdeploy/vision/perception/paddle3d/smoke.py diff --git a/python/fastdeploy/vision/segmentation/__init__.py b/bindings/python/fastdeploy/vision/segmentation/__init__.py similarity index 100% rename from python/fastdeploy/vision/segmentation/__init__.py rename to bindings/python/fastdeploy/vision/segmentation/__init__.py diff --git a/python/fastdeploy/vision/segmentation/ppseg/__init__.py b/bindings/python/fastdeploy/vision/segmentation/ppseg/__init__.py similarity index 100% rename from python/fastdeploy/vision/segmentation/ppseg/__init__.py rename to bindings/python/fastdeploy/vision/segmentation/ppseg/__init__.py diff --git a/python/fastdeploy/vision/sr/__init__.py b/bindings/python/fastdeploy/vision/sr/__init__.py similarity index 100% rename from python/fastdeploy/vision/sr/__init__.py rename to bindings/python/fastdeploy/vision/sr/__init__.py diff --git a/python/fastdeploy/vision/sr/ppsr/__init__.py b/bindings/python/fastdeploy/vision/sr/ppsr/__init__.py similarity index 100% rename from python/fastdeploy/vision/sr/ppsr/__init__.py rename to bindings/python/fastdeploy/vision/sr/ppsr/__init__.py diff --git a/python/fastdeploy/vision/tracking/__init__.py b/bindings/python/fastdeploy/vision/tracking/__init__.py similarity index 100% rename from python/fastdeploy/vision/tracking/__init__.py rename to bindings/python/fastdeploy/vision/tracking/__init__.py diff --git a/python/fastdeploy/vision/tracking/pptracking/__init__.py b/bindings/python/fastdeploy/vision/tracking/pptracking/__init__.py similarity index 100% rename from python/fastdeploy/vision/tracking/pptracking/__init__.py rename to bindings/python/fastdeploy/vision/tracking/pptracking/__init__.py diff --git a/python/fastdeploy/vision/utils.py b/bindings/python/fastdeploy/vision/utils.py similarity index 100% rename from python/fastdeploy/vision/utils.py rename to bindings/python/fastdeploy/vision/utils.py diff --git a/python/fastdeploy/vision/visualize/__init__.py b/bindings/python/fastdeploy/vision/visualize/__init__.py similarity index 100% rename from python/fastdeploy/vision/visualize/__init__.py rename to bindings/python/fastdeploy/vision/visualize/__init__.py diff --git a/python/requirements.txt b/bindings/python/requirements.txt similarity index 100% rename from python/requirements.txt rename to bindings/python/requirements.txt diff --git a/python/scripts/__init__.py b/bindings/python/scripts/__init__.py similarity index 100% rename from python/scripts/__init__.py rename to bindings/python/scripts/__init__.py diff --git a/python/scripts/build_gpu.sh b/bindings/python/scripts/build_gpu.sh similarity index 100% rename from python/scripts/build_gpu.sh rename to bindings/python/scripts/build_gpu.sh diff --git a/python/scripts/process_libraries.py.in b/bindings/python/scripts/process_libraries.py.in similarity index 100% rename from python/scripts/process_libraries.py.in rename to bindings/python/scripts/process_libraries.py.in diff --git a/python/setup.py b/bindings/python/setup.py similarity index 94% rename from python/setup.py rename to bindings/python/setup.py index 1248c3a8d41..68111373fe8 100755 --- a/python/setup.py +++ b/bindings/python/setup.py @@ -20,8 +20,8 @@ import shutil import os -TOP_DIR = os.path.realpath(os.path.dirname(__file__)) -TOP_DIR = os.path.split(TOP_DIR)[0] +CURRENT_DIR = os.path.realpath(os.path.dirname(__file__)) +TOP_DIR = os.path.dirname(os.path.dirname(CURRENT_DIR)) PACKAGE_NAME = os.getenv("PACKAGE_NAME", "fastdeploy") wheel_name = "fastdeploy-python" @@ -42,7 +42,7 @@ from textwrap import dedent import multiprocessing -with open(os.path.join(TOP_DIR, "python", "requirements.txt")) as fin: +with open(os.path.join(CURRENT_DIR, "requirements.txt")) as fin: REQUIRED_PACKAGES = fin.read() if os.getenv("BUILD_ON_CPU", "OFF") == "ON": @@ -114,8 +114,8 @@ setup_configs["CMAKE_CXX_COMPILER"] = os.getenv("CMAKE_CXX_COMPILER") SRC_DIR = os.path.join(TOP_DIR, PACKAGE_NAME) -PYTHON_SRC_DIR = os.path.join(TOP_DIR, "python", PACKAGE_NAME) -CMAKE_BUILD_DIR = os.path.join(TOP_DIR, 'python', '.setuptools-cmake-build') +PYTHON_SRC_DIR = os.path.join(CURRENT_DIR, PACKAGE_NAME) +CMAKE_BUILD_DIR = os.path.join(CURRENT_DIR, '.setuptools-cmake-build') WINDOWS = (os.name == 'nt') @@ -148,16 +148,21 @@ if setup_configs["PADDLEINFERENCE_VERSION"] != "": extra_version_info += ("." + setup_configs["PADDLEINFERENCE_VERSION"]) -with open(os.path.join(TOP_DIR, 'VERSION_NUMBER')) as version_file: - VersionInfo = namedtuple('VersionInfo', [ - 'version', 'git_version', 'extra_version_info', 'enable_trt_backend', - 'enable_paddle_backend', 'WITH_CUDA' - ])(version=version_file.read().strip(), - git_version=git_version, - extra_version_info=extra_version_info.strip("."), - enable_trt_backend=setup_configs["ENABLE_TRT_BACKEND"], - enable_paddle_backend=setup_configs["ENABLE_PADDLE_BACKEND"], - WITH_CUDA=setup_configs["WITH_CUDA"]) +version_str = os.getenv("FASTDEPLOY_VERSION", "0.0.0") +version_file_path = os.path.join(TOP_DIR, 'VERSION_NUMBER') +if os.path.exists(version_file_path): + with open(version_file_path) as vf: + version_str = vf.read().strip() + +VersionInfo = namedtuple('VersionInfo', [ + 'version', 'git_version', 'extra_version_info', 'enable_trt_backend', + 'enable_paddle_backend', 'WITH_CUDA' +])(version=version_str, + git_version=git_version, + extra_version_info=extra_version_info.strip("."), + enable_trt_backend=setup_configs["ENABLE_TRT_BACKEND"], + enable_paddle_backend=setup_configs["ENABLE_PADDLE_BACKEND"], + WITH_CUDA=setup_configs["WITH_CUDA"]) ################################################################################ # Pre Check @@ -407,12 +412,12 @@ def run(self): if sys.argv[1] == "install" or sys.argv[1] == "bdist_wheel": shutil.copy( - os.path.join(TOP_DIR, "ThirdPartyNotices.txt"), + os.path.join(TOP_DIR, "docs", "ThirdPartyNotices.txt"), os.path.join(TOP_DIR, PACKAGE_NAME)) shutil.copy( os.path.join(TOP_DIR, "LICENSE"), os.path.join(TOP_DIR, PACKAGE_NAME)) if not os.path.exists( - os.path.join(TOP_DIR, "python", "fastdeploy", "libs", + os.path.join(CURRENT_DIR, "fastdeploy", "libs", "third_libs")): print( "Didn't detect path: fastdeploy/libs/third_libs exist, please execute `python setup.py build` first" diff --git a/FastDeploy.cmake.in b/cmake/FastDeploy.cmake.in similarity index 100% rename from FastDeploy.cmake.in rename to cmake/FastDeploy.cmake.in diff --git a/FastDeployCSharp.cmake.in b/cmake/FastDeployCSharp.cmake.in similarity index 100% rename from FastDeployCSharp.cmake.in rename to cmake/FastDeployCSharp.cmake.in diff --git a/cmake/FindEigen3.cmake b/cmake/FindEigen3.cmake new file mode 100644 index 00000000000..04810ddee1a --- /dev/null +++ b/cmake/FindEigen3.cmake @@ -0,0 +1,33 @@ +# Try modern CMake Config mode first (macOS Homebrew, Linux apt, vcpkg) +find_package(Eigen3 ${Eigen3_FIND_VERSION} CONFIG QUIET) +if(TARGET Eigen3::Eigen OR Eigen3_FOUND) + set(Eigen3_FOUND TRUE) + set(EIGEN3_FOUND TRUE) + return() +endif() + +# Fallback: Find in-tree third_party/eigen or prefix path +find_path(EIGEN3_INCLUDE_DIR NAMES signature_of_eigen3_matrix_library + HINTS + ENV EIGEN3_ROOT + ENV EIGEN3_ROOT_DIR + "${CMAKE_CURRENT_SOURCE_DIR}/third_party/eigen" + PATHS + ${CMAKE_PREFIX_PATH} + ${CMAKE_INSTALL_PREFIX}/include + PATH_SUFFIXES eigen3 eigen +) + +if(EIGEN3_INCLUDE_DIR) + set(EIGEN3_FOUND TRUE) + set(Eigen3_FOUND TRUE) + if(NOT TARGET Eigen3::Eigen) + add_library(Eigen3::Eigen INTERFACE IMPORTED) + set_target_properties(Eigen3::Eigen PROPERTIES + INTERFACE_INCLUDE_DIRECTORIES "${EIGEN3_INCLUDE_DIR}") + endif() +endif() + +include(FindPackageHandleStandardArgs) +find_package_handle_standard_args(Eigen3 DEFAULT_MSG EIGEN3_INCLUDE_DIR) + diff --git a/cmake/build_paddle2onnx.cmake b/cmake/build_paddle2onnx.cmake deleted file mode 100644 index c7cc152467f..00000000000 --- a/cmake/build_paddle2onnx.cmake +++ /dev/null @@ -1,41 +0,0 @@ -add_definitions(-DMAX_ONNX_OPSET_VERSION=16) -add_definitions(-DPADDLE2ONNX_LIB) - -# Third dependency: onnx -if(NOT TARGET onnx_proto) - if(NOT ONNX_NAMESPACE) - set(ONNX_NAMESPACE "paddle2onnx") - endif() - add_definitions("-DONNX_NAMESPACE=${ONNX_NAMESPACE}") - - set(MSVC_STATIC_CRT ON) - if(ONNX_CUSTOM_PROTOC_PATH) - if(WIN32) - if(MSVC_STATIC_CRT) - # MT - set(ONNX_USE_MSVC_STATIC_RUNTIME ON) - else() - # MD - set(ONNX_USE_MSVC_STATIC_RUNTIME OFF) - endif() - set(ONNX_CUSTOM_PROTOC_PATH "${ONNX_CUSTOM_PROTOC_PATH};$ENV{PATH}") - else() - set(ONNX_CUSTOM_PROTOC_PATH "${ONNX_CUSTOM_PROTOC_PATH}:$ENV{PATH}") - endif() - set(ENV{PATH} ${ONNX_CUSTOM_PROTOC_PATH}) - endif() - - set(CMAKE_POSITION_INDEPENDENT_CODE ON) - add_subdirectory(${PROJECT_SOURCE_DIR}/third_party/onnx) -endif() - -include_directories(${PROJECT_SOURCE_DIR}) -include_directories(${CMAKE_CURRENT_BINARY_DIR}) -include_directories(${CMAKE_CURRENT_BINARY_DIR}/third_party/onnx) - -include_directories(${PROJECT_SOURCE_DIR}/third_party/optimizer) -add_subdirectory(${PROJECT_SOURCE_DIR}/paddle2onnx/proto) - -file(GLOB_RECURSE PADDLE2ONNX_ALL_SRCS ${PROJECT_SOURCE_DIR}/paddle2onnx/*.cc ${PROJECT_SOURCE_DIR}/third_party/optimizer/onnxoptimizer/*.cc) -list(REMOVE_ITEM PADDLE2ONNX_ALL_SRCS ${PROJECT_SOURCE_DIR}/paddle2onnx/cpp2py_export.cc ${PROJECT_SOURCE_DIR}/third_party/optimizer/onnxoptimizer/cpp2py_export.cc) - diff --git a/cmake/config_cpack.cmake b/cmake/config_cpack.cmake index 7204d620d5b..284356d9231 100644 --- a/cmake/config_cpack.cmake +++ b/cmake/config_cpack.cmake @@ -23,15 +23,15 @@ set(CPACK_PACKAGE_FILE_NAME "${PROJECT_NAME}-${PACKAGE_SYS_VERSION}-${FASTDEPLOY set(CPACK_PACKAGE_NAME "${PROJECT_NAME}") set(CPACK_DEBIAN_PACKAGE_CONTROL_STRICT_PERMISSION TRUE) -configure_file(cpack/debian_postinst.in cpack/postinst @ONLY) -configure_file(cpack/debian_prerm.in cpack/prerm @ONLY) +configure_file(${CMAKE_CURRENT_LIST_DIR}/cpack/debian_postinst.in ${CMAKE_CURRENT_BINARY_DIR}/cpack/postinst @ONLY) +configure_file(${CMAKE_CURRENT_LIST_DIR}/cpack/debian_prerm.in ${CMAKE_CURRENT_BINARY_DIR}/cpack/prerm @ONLY) set(CPACK_DEBIAN_PACKAGE_CONTROL_EXTRA "${CMAKE_CURRENT_BINARY_DIR}/cpack/postinst" "${CMAKE_CURRENT_BINARY_DIR}/cpack/prerm") set(CPACK_RPM_PACKAGE_AUTOREQ FALSE) -configure_file(cpack/rpm_postinst.in cpack/rpm_postinst @ONLY) -configure_file(cpack/rpm_postrm.in cpack/rpm_postrm @ONLY) +configure_file(${CMAKE_CURRENT_LIST_DIR}/cpack/rpm_postinst.in ${CMAKE_CURRENT_BINARY_DIR}/cpack/rpm_postinst @ONLY) +configure_file(${CMAKE_CURRENT_LIST_DIR}/cpack/rpm_postrm.in ${CMAKE_CURRENT_BINARY_DIR}/cpack/rpm_postrm @ONLY) set(CPACK_RPM_POST_INSTALL_SCRIPT_FILE "${CMAKE_CURRENT_BINARY_DIR}/cpack/rpm_postinst") set(CPACK_RPM_POST_UNINSTALL_SCRIPT_FILE "${CMAKE_CURRENT_BINARY_DIR}/cpack/rpm_postrm") diff --git a/cpack/debian_postinst.in b/cmake/cpack/debian_postinst.in similarity index 100% rename from cpack/debian_postinst.in rename to cmake/cpack/debian_postinst.in diff --git a/cpack/debian_prerm.in b/cmake/cpack/debian_prerm.in similarity index 100% rename from cpack/debian_prerm.in rename to cmake/cpack/debian_prerm.in diff --git a/cpack/rpm_postinst.in b/cmake/cpack/rpm_postinst.in similarity index 100% rename from cpack/rpm_postinst.in rename to cmake/cpack/rpm_postinst.in diff --git a/cpack/rpm_postrm.in b/cmake/cpack/rpm_postrm.in similarity index 100% rename from cpack/rpm_postrm.in rename to cmake/cpack/rpm_postrm.in diff --git a/cmake/onnxruntime.cmake b/cmake/onnxruntime.cmake index 31c133e888b..2bc32bd7b54 100644 --- a/cmake/onnxruntime.cmake +++ b/cmake/onnxruntime.cmake @@ -41,7 +41,7 @@ else() endif() set(CMAKE_BUILD_RPATH "${CMAKE_BUILD_RPATH}" "${ONNXRUNTIME_LIB_DIR}") -set(ONNXRUNTIME_VERSION "1.12.0") +set(ONNXRUNTIME_VERSION "1.29.0") set(ONNXRUNTIME_URL_PREFIX "https://bj.bcebos.com/paddle2onnx/libs/") if(WIN32) diff --git a/cmake/poros.cmake b/cmake/poros.cmake deleted file mode 100755 index b2cae46657d..00000000000 --- a/cmake/poros.cmake +++ /dev/null @@ -1,95 +0,0 @@ -# Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -include(ExternalProject) - -if(NOT ENABLE_TRT_BACKEND) - message(FATAL_ERROR "While ENABLE_POROS_BACKEND, requires ENABLE_TRT_BACKEND=ON, but now its OFF.") -endif() - -set(POROS_PROJECT "extern_poros") -set(POROS_PREFIX_DIR ${THIRD_PARTY_PATH}/poros) -set(POROS_SOURCE_DIR - ${THIRD_PARTY_PATH}/poros/src/${POROS_PROJECT}) -set(POROS_INSTALL_DIR ${THIRD_PARTY_PATH}/install/poros) -set(POROS_INC_DIR - "${POROS_INSTALL_DIR}/include" - CACHE PATH "poros include directory." FORCE) -set(POROS_LIB_DIR - "${POROS_INSTALL_DIR}/lib/" - CACHE PATH "poros lib directory." FORCE) -set(CMAKE_BUILD_RPATH "${CMAKE_BUILD_RPATH}" - "${POROS_LIB_DIR}") - -include_directories(${POROS_INC_DIR}) -if(WIN32) - message(FATAL_ERROR "Poros Backend doesn't support Windows now.") -elseif(APPLE) - message(FATAL_ERROR "Poros Backend doesn't support Mac OSX now.") -else() - set(POROS_COMPILE_LIB - "${POROS_INSTALL_DIR}/lib/libporos.so" - CACHE FILEPATH "poros compile library." FORCE) -endif(WIN32) - -set(POROS_URL_BASE "https://bj.bcebos.com/fastdeploy/third_libs/") -set(POROS_VERSION "0.1.0") -if(WIN32) - message(FATAL_ERROR "Poros Backend doesn't support Windows now.") -elseif(APPLE) - message(FATAL_ERROR "Poros Backend doesn't support Mac OSX now.") -else() - if(CMAKE_HOST_SYSTEM_PROCESSOR MATCHES "aarch64") - message(FATAL_ERROR "Poros Backend doesn't support linux aarch64 now.") - else() - if(WITH_CUDA) - set(POROS_FILE "poros_manylinux_torch1.12.1_cu116_trt8.4_gcc82-${POROS_VERSION}.tar.gz") - else() - message(FATAL_ERROR "Poros currently only provides precompiled packages for the GPU version.") - endif() - endif() -endif() -set(POROS_URL "${POROS_URL_BASE}${POROS_FILE}") - -ExternalProject_Add( - ${POROS_PROJECT} - ${EXTERNAL_PROJECT_LOG_ARGS} - URL ${POROS_URL} - PREFIX ${POROS_PREFIX_DIR} - DOWNLOAD_NO_PROGRESS 1 - CONFIGURE_COMMAND "" - BUILD_COMMAND "" - UPDATE_COMMAND "" - INSTALL_COMMAND - ${CMAKE_COMMAND} -E copy_directory ${POROS_SOURCE_DIR} ${POROS_INSTALL_DIR} - BUILD_BYPRODUCTS ${POROS_COMPILE_LIB}) - -add_library(external_poros STATIC IMPORTED GLOBAL) -set_property(TARGET external_poros PROPERTY IMPORTED_LOCATION - ${POROS_COMPILE_LIB}) -add_dependencies(external_poros ${POROS_PROJECT}) - -# Download libtorch.so with ABI=1 -set(TORCH_URL_BASE "https://bj.bcebos.com/fastdeploy/third_libs/") -set(TORCH_FILE "libtorch-cxx11-abi-shared-with-deps-1.12.1-cu116.zip") -set(TORCH_URL "${TORCH_URL_BASE}${TORCH_FILE}") -message(STATUS "Use the default Torch lib from: ${TORCH_URL}") -download_and_decompress(${TORCH_URL} ${CMAKE_CURRENT_BINARY_DIR}/${TORCH_FILE} ${THIRD_PARTY_PATH}/install) -if(EXISTS ${THIRD_PARTY_PATH}/install/torch) - file(REMOVE_RECURSE ${THIRD_PARTY_PATH}/install/torch) -endif() -file(RENAME ${THIRD_PARTY_PATH}/install/libtorch/ ${THIRD_PARTY_PATH}/install/torch) -set(TORCH_INCLUDE_DIRS ${THIRD_PARTY_PATH}/install/torch/include) -find_library(TORCH_LIBRARY torch ${THIRD_PARTY_PATH}/install/torch/lib NO_DEFAULT_PATH) -include_directories(${TORCH_INCLUDE_DIRS}) -list(APPEND DEPEND_LIBS ${TORCH_LIBRARY}) diff --git a/ThirdPartyNotices.txt b/docs/ThirdPartyNotices.txt similarity index 100% rename from ThirdPartyNotices.txt rename to docs/ThirdPartyNotices.txt diff --git a/README_CN.md b/docs/legacy/README_CN.md similarity index 97% rename from README_CN.md rename to docs/legacy/README_CN.md index a5b4ad29793..0a49ea920e8 100755 --- a/README_CN.md +++ b/docs/legacy/README_CN.md @@ -57,16 +57,11 @@ - FastDeploy系列[**直播课程回放**](https://aistudio.baidu.com/aistudio/education/group/info/27800) -- **2023.01.17** 发布 [**YOLOv8**](./examples/vision/detection/paddledetection/) 在FastDeploy系列硬件的部署支持。 其中包括 [**Paddle YOLOv8**](https://github.com/PaddlePaddle/PaddleYOLO/tree/release/2.5/configs/yolov8) 以及 [**社区 ultralytics YOLOv8**](https://github.com/ultralytics/ultralytics) - - [**Paddle YOLOv8**](https://github.com/PaddlePaddle/PaddleYOLO/tree/release/2.5/configs/yolov8) 可以部署的硬件:[**Intel CPU**](./examples/vision/detection/paddledetection/python/infer_yolov8.py)、[**NVIDIA GPU**](./examples/vision/detection/paddledetection/python/infer_yolov8.py)、[**Jetson**](./examples/vision/detection/paddledetection/python/infer_yolov8.py)、[**飞腾**](./examples/vision/detection/paddledetection/python/infer_yolov8.py)、[**昆仑芯**](./examples/vision/detection/paddledetection/python/infer_yolov8.py)、[**昇腾**](./examples/vision/detection/paddledetection/python/infer_yolov8.py)、[**ARM CPU**](./examples/vision/detection/paddledetection/cpp/infer_yolov8.cc)、[**RK3588**](./examples/vision/detection/paddledetection/rknpu2) 和 [**Sophgo TPU**](./examples/vision/detection/paddledetection/sophgo), 部分硬件包含 **Python** 部署和 **C++** 部署; - - [**社区 ultralytics YOLOv8**](https://github.com/ultralytics/ultralytics) 可以部署的硬件:[**Intel CPU**](./examples/vision/detection/yolov8)、[**NVIDIA GPU**](./examples/vision/detection/yolov8)、[**Jetson**](./examples/vision/detection/yolov8),均包含 **Python** 部署和 **C++** 部署; - - FastDeploy 一行模型API切换,可以实现**YOLOv8**、 **PP-YOLOE+**、**YOLOv5** 等模型性能对比。 - - 服务化部署结合VisualDL新增支持可视化部署。在FastDeploy容器中启动VDL服务后,即可在VDL界面修改模型配置、启动/管理模型服务、查看性能数据、发送请求等,详细操作可参考相关文档 +- 服务化部署结合VisualDL新增支持可视化部署。在FastDeploy容器中启动VDL服务后,即可在VDL界面修改模型配置、启动/管理模型服务、查看性能数据、发送请求等,详细操作可参考相关文档 - [Serving可视化部署](https://github.com/PaddlePaddle/FastDeploy/blob/develop/serving/docs/zh_CN/vdl_management.md) - [Serving可视化请求](https://github.com/PaddlePaddle/FastDeploy/blob/develop/serving/docs/zh_CN/client.md#%E4%BD%BF%E7%94%A8fastdeploy-client%E8%BF%9B%E8%A1%8C%E5%8F%AF%E8%A7%86%E5%8C%96%E8%AF%B7%E6%B1%82) - - **✨👥✨ 社区交流** - **Slack**:Join our [Slack community](https://join.slack.com/t/fastdeployworkspace/shared_invite/zt-1o50e4voz-zbiIneCNRf_eH99eS2NVLg) and chat with other community members about ideas diff --git a/README_EN.md b/docs/legacy/README_EN.md similarity index 100% rename from README_EN.md rename to docs/legacy/README_EN.md diff --git a/examples/text/uie/cpp/CMakeLists.txt b/examples/text/uie/cpp/CMakeLists.txt index e74df6224cb..925804aeaeb 100644 --- a/examples/text/uie/cpp/CMakeLists.txt +++ b/examples/text/uie/cpp/CMakeLists.txt @@ -14,6 +14,9 @@ PROJECT(infer_demo C CXX) CMAKE_MINIMUM_REQUIRED (VERSION 3.10) +if(MSVC) + add_definitions(/utf-8) +endif() option(FASTDEPLOY_INSTALL_DIR "Path of downloaded fastdeploy sdk.") diff --git a/examples/text/uie/cpp/infer.cc b/examples/text/uie/cpp/infer.cc index 5c1bb704ddc..95b21973326 100644 --- a/examples/text/uie/cpp/infer.cc +++ b/examples/text/uie/cpp/infer.cc @@ -17,6 +17,10 @@ #include "fastdeploy/text.h" +#ifdef WIN32 +#include +#endif + using namespace paddlenlp; #ifdef WIN32 @@ -77,6 +81,9 @@ int main(int argc, char* argv[]) { predictor.Predict({"2月8日上午北京冬奥会自由式滑雪女子大跳台决赛中中国选手谷" "爱凌以188.25分获得金牌!"}, &results); +#ifdef WIN32 + SetConsoleOutputCP(CP_UTF8); +#endif std::cout << results << std::endl; results.clear(); diff --git a/tutorials/README.md b/examples/tutorials/README.md similarity index 100% rename from tutorials/README.md rename to examples/tutorials/README.md diff --git a/tutorials/README_CN.md b/examples/tutorials/README_CN.md similarity index 100% rename from tutorials/README_CN.md rename to examples/tutorials/README_CN.md diff --git a/tutorials/encrypt_model/README.md b/examples/tutorials/encrypt_model/README.md similarity index 100% rename from tutorials/encrypt_model/README.md rename to examples/tutorials/encrypt_model/README.md diff --git a/tutorials/encrypt_model/README_CN.md b/examples/tutorials/encrypt_model/README_CN.md similarity index 100% rename from tutorials/encrypt_model/README_CN.md rename to examples/tutorials/encrypt_model/README_CN.md diff --git a/tutorials/encrypt_model/encrypt.py b/examples/tutorials/encrypt_model/encrypt.py similarity index 100% rename from tutorials/encrypt_model/encrypt.py rename to examples/tutorials/encrypt_model/encrypt.py diff --git a/tutorials/image_decoder/README.md b/examples/tutorials/image_decoder/README.md similarity index 100% rename from tutorials/image_decoder/README.md rename to examples/tutorials/image_decoder/README.md diff --git a/tutorials/image_decoder/README_CN.md b/examples/tutorials/image_decoder/README_CN.md similarity index 100% rename from tutorials/image_decoder/README_CN.md rename to examples/tutorials/image_decoder/README_CN.md diff --git a/tutorials/image_decoder/cpp/CMakeLists.txt b/examples/tutorials/image_decoder/cpp/CMakeLists.txt similarity index 100% rename from tutorials/image_decoder/cpp/CMakeLists.txt rename to examples/tutorials/image_decoder/cpp/CMakeLists.txt diff --git a/tutorials/image_decoder/cpp/README.md b/examples/tutorials/image_decoder/cpp/README.md similarity index 100% rename from tutorials/image_decoder/cpp/README.md rename to examples/tutorials/image_decoder/cpp/README.md diff --git a/tutorials/image_decoder/cpp/README_CN.md b/examples/tutorials/image_decoder/cpp/README_CN.md similarity index 100% rename from tutorials/image_decoder/cpp/README_CN.md rename to examples/tutorials/image_decoder/cpp/README_CN.md diff --git a/tutorials/image_decoder/cpp/main.cc b/examples/tutorials/image_decoder/cpp/main.cc similarity index 100% rename from tutorials/image_decoder/cpp/main.cc rename to examples/tutorials/image_decoder/cpp/main.cc diff --git a/tutorials/intel_gpu/README.md b/examples/tutorials/intel_gpu/README.md similarity index 100% rename from tutorials/intel_gpu/README.md rename to examples/tutorials/intel_gpu/README.md diff --git a/tutorials/intel_gpu/README_CN.md b/examples/tutorials/intel_gpu/README_CN.md similarity index 100% rename from tutorials/intel_gpu/README_CN.md rename to examples/tutorials/intel_gpu/README_CN.md diff --git a/tutorials/intel_gpu/cpp/CMakeLists.txt b/examples/tutorials/intel_gpu/cpp/CMakeLists.txt similarity index 100% rename from tutorials/intel_gpu/cpp/CMakeLists.txt rename to examples/tutorials/intel_gpu/cpp/CMakeLists.txt diff --git a/tutorials/intel_gpu/cpp/README.md b/examples/tutorials/intel_gpu/cpp/README.md similarity index 100% rename from tutorials/intel_gpu/cpp/README.md rename to examples/tutorials/intel_gpu/cpp/README.md diff --git a/tutorials/intel_gpu/cpp/README_CN.md b/examples/tutorials/intel_gpu/cpp/README_CN.md similarity index 100% rename from tutorials/intel_gpu/cpp/README_CN.md rename to examples/tutorials/intel_gpu/cpp/README_CN.md diff --git a/tutorials/intel_gpu/cpp/infer_ppyoloe.cc b/examples/tutorials/intel_gpu/cpp/infer_ppyoloe.cc similarity index 100% rename from tutorials/intel_gpu/cpp/infer_ppyoloe.cc rename to examples/tutorials/intel_gpu/cpp/infer_ppyoloe.cc diff --git a/tutorials/intel_gpu/cpp/infer_resnet50.cc b/examples/tutorials/intel_gpu/cpp/infer_resnet50.cc similarity index 100% rename from tutorials/intel_gpu/cpp/infer_resnet50.cc rename to examples/tutorials/intel_gpu/cpp/infer_resnet50.cc diff --git a/tutorials/intel_gpu/python/README.md b/examples/tutorials/intel_gpu/python/README.md similarity index 100% rename from tutorials/intel_gpu/python/README.md rename to examples/tutorials/intel_gpu/python/README.md diff --git a/tutorials/intel_gpu/python/README_CN.md b/examples/tutorials/intel_gpu/python/README_CN.md similarity index 100% rename from tutorials/intel_gpu/python/README_CN.md rename to examples/tutorials/intel_gpu/python/README_CN.md diff --git a/tutorials/intel_gpu/python/infer_ppyoloe.py b/examples/tutorials/intel_gpu/python/infer_ppyoloe.py similarity index 100% rename from tutorials/intel_gpu/python/infer_ppyoloe.py rename to examples/tutorials/intel_gpu/python/infer_ppyoloe.py diff --git a/tutorials/intel_gpu/python/infer_resnet50.py b/examples/tutorials/intel_gpu/python/infer_resnet50.py similarity index 100% rename from tutorials/intel_gpu/python/infer_resnet50.py rename to examples/tutorials/intel_gpu/python/infer_resnet50.py diff --git a/tutorials/multi_thread/README.md b/examples/tutorials/multi_thread/README.md similarity index 100% rename from tutorials/multi_thread/README.md rename to examples/tutorials/multi_thread/README.md diff --git a/tutorials/multi_thread/README_CN.md b/examples/tutorials/multi_thread/README_CN.md similarity index 100% rename from tutorials/multi_thread/README_CN.md rename to examples/tutorials/multi_thread/README_CN.md diff --git a/tutorials/multi_thread/cpp/pipeline/CMakeLists.txt b/examples/tutorials/multi_thread/cpp/pipeline/CMakeLists.txt similarity index 100% rename from tutorials/multi_thread/cpp/pipeline/CMakeLists.txt rename to examples/tutorials/multi_thread/cpp/pipeline/CMakeLists.txt diff --git a/tutorials/multi_thread/cpp/pipeline/README.md b/examples/tutorials/multi_thread/cpp/pipeline/README.md similarity index 100% rename from tutorials/multi_thread/cpp/pipeline/README.md rename to examples/tutorials/multi_thread/cpp/pipeline/README.md diff --git a/tutorials/multi_thread/cpp/pipeline/README_CN.md b/examples/tutorials/multi_thread/cpp/pipeline/README_CN.md similarity index 100% rename from tutorials/multi_thread/cpp/pipeline/README_CN.md rename to examples/tutorials/multi_thread/cpp/pipeline/README_CN.md diff --git a/tutorials/multi_thread/cpp/pipeline/multi_thread_ocr.cc b/examples/tutorials/multi_thread/cpp/pipeline/multi_thread_ocr.cc similarity index 100% rename from tutorials/multi_thread/cpp/pipeline/multi_thread_ocr.cc rename to examples/tutorials/multi_thread/cpp/pipeline/multi_thread_ocr.cc diff --git a/tutorials/multi_thread/cpp/single_model/CMakeLists.txt b/examples/tutorials/multi_thread/cpp/single_model/CMakeLists.txt similarity index 100% rename from tutorials/multi_thread/cpp/single_model/CMakeLists.txt rename to examples/tutorials/multi_thread/cpp/single_model/CMakeLists.txt diff --git a/tutorials/multi_thread/cpp/single_model/README.md b/examples/tutorials/multi_thread/cpp/single_model/README.md similarity index 100% rename from tutorials/multi_thread/cpp/single_model/README.md rename to examples/tutorials/multi_thread/cpp/single_model/README.md diff --git a/tutorials/multi_thread/cpp/single_model/README_CN.md b/examples/tutorials/multi_thread/cpp/single_model/README_CN.md similarity index 100% rename from tutorials/multi_thread/cpp/single_model/README_CN.md rename to examples/tutorials/multi_thread/cpp/single_model/README_CN.md diff --git a/tutorials/multi_thread/cpp/single_model/multi_thread.cc b/examples/tutorials/multi_thread/cpp/single_model/multi_thread.cc similarity index 100% rename from tutorials/multi_thread/cpp/single_model/multi_thread.cc rename to examples/tutorials/multi_thread/cpp/single_model/multi_thread.cc diff --git a/tutorials/multi_thread/python/pipeline/README.md b/examples/tutorials/multi_thread/python/pipeline/README.md similarity index 100% rename from tutorials/multi_thread/python/pipeline/README.md rename to examples/tutorials/multi_thread/python/pipeline/README.md diff --git a/tutorials/multi_thread/python/pipeline/README_CN.md b/examples/tutorials/multi_thread/python/pipeline/README_CN.md similarity index 100% rename from tutorials/multi_thread/python/pipeline/README_CN.md rename to examples/tutorials/multi_thread/python/pipeline/README_CN.md diff --git a/tutorials/multi_thread/python/pipeline/multi_thread_process_ocr.py b/examples/tutorials/multi_thread/python/pipeline/multi_thread_process_ocr.py similarity index 100% rename from tutorials/multi_thread/python/pipeline/multi_thread_process_ocr.py rename to examples/tutorials/multi_thread/python/pipeline/multi_thread_process_ocr.py diff --git a/tutorials/multi_thread/python/single_model/README.md b/examples/tutorials/multi_thread/python/single_model/README.md similarity index 100% rename from tutorials/multi_thread/python/single_model/README.md rename to examples/tutorials/multi_thread/python/single_model/README.md diff --git a/tutorials/multi_thread/python/single_model/README_CN.md b/examples/tutorials/multi_thread/python/single_model/README_CN.md similarity index 100% rename from tutorials/multi_thread/python/single_model/README_CN.md rename to examples/tutorials/multi_thread/python/single_model/README_CN.md diff --git a/tutorials/multi_thread/python/single_model/multi_thread_process.py b/examples/tutorials/multi_thread/python/single_model/multi_thread_process.py similarity index 100% rename from tutorials/multi_thread/python/single_model/multi_thread_process.py rename to examples/tutorials/multi_thread/python/single_model/multi_thread_process.py diff --git a/tutorials/use_c_csharp_sdk/README.md b/examples/tutorials/use_c_csharp_sdk/README.md similarity index 100% rename from tutorials/use_c_csharp_sdk/README.md rename to examples/tutorials/use_c_csharp_sdk/README.md diff --git a/tutorials/use_c_csharp_sdk/README_CN.md b/examples/tutorials/use_c_csharp_sdk/README_CN.md similarity index 100% rename from tutorials/use_c_csharp_sdk/README_CN.md rename to examples/tutorials/use_c_csharp_sdk/README_CN.md diff --git a/tutorials/vision_processor/README.md b/examples/tutorials/vision_processor/README.md similarity index 100% rename from tutorials/vision_processor/README.md rename to examples/tutorials/vision_processor/README.md diff --git a/tutorials/vision_processor/README_CN.md b/examples/tutorials/vision_processor/README_CN.md similarity index 100% rename from tutorials/vision_processor/README_CN.md rename to examples/tutorials/vision_processor/README_CN.md diff --git a/tutorials/vision_processor/cpp/CMakeLists.txt b/examples/tutorials/vision_processor/cpp/CMakeLists.txt similarity index 100% rename from tutorials/vision_processor/cpp/CMakeLists.txt rename to examples/tutorials/vision_processor/cpp/CMakeLists.txt diff --git a/tutorials/vision_processor/cpp/README.md b/examples/tutorials/vision_processor/cpp/README.md similarity index 100% rename from tutorials/vision_processor/cpp/README.md rename to examples/tutorials/vision_processor/cpp/README.md diff --git a/tutorials/vision_processor/cpp/README_CN.md b/examples/tutorials/vision_processor/cpp/README_CN.md similarity index 100% rename from tutorials/vision_processor/cpp/README_CN.md rename to examples/tutorials/vision_processor/cpp/README_CN.md diff --git a/tutorials/vision_processor/cpp/main.cc b/examples/tutorials/vision_processor/cpp/main.cc similarity index 100% rename from tutorials/vision_processor/cpp/main.cc rename to examples/tutorials/vision_processor/cpp/main.cc diff --git a/tutorials/vision_processor/python/README.md b/examples/tutorials/vision_processor/python/README.md similarity index 100% rename from tutorials/vision_processor/python/README.md rename to examples/tutorials/vision_processor/python/README.md diff --git a/tutorials/vision_processor/python/README_CN.md b/examples/tutorials/vision_processor/python/README_CN.md similarity index 100% rename from tutorials/vision_processor/python/README_CN.md rename to examples/tutorials/vision_processor/python/README_CN.md diff --git a/tutorials/vision_processor/python/preprocess.py b/examples/tutorials/vision_processor/python/preprocess.py similarity index 100% rename from tutorials/vision_processor/python/preprocess.py rename to examples/tutorials/vision_processor/python/preprocess.py diff --git a/examples/vision/ocr/PP-OCR/cpu-gpu/cpp/README.md b/examples/vision/ocr/PP-OCR/cpu-gpu/cpp/README.md index b0e735e724f..48da334120e 100644 --- a/examples/vision/ocr/PP-OCR/cpu-gpu/cpp/README.md +++ b/examples/vision/ocr/PP-OCR/cpu-gpu/cpp/README.md @@ -62,7 +62,7 @@ wget https://gitee.com/paddlepaddle/PaddleOCR/raw/release/2.6/ppocr/utils/dict/l # 运行部署示例 # 在CPU上使用Paddle Inference推理 ./infer_demo ./ch_PP-OCRv3_det_infer ./ch_ppocr_mobile_v2.0_cls_infer ./ch_PP-OCRv3_rec_infer ./ppocr_keys_v1.txt ./12.jpg 0 -# 在CPU上使用OenVINO推理 +# 在CPU上使用OpenVINO推理 ./infer_demo ./ch_PP-OCRv3_det_infer ./ch_ppocr_mobile_v2.0_cls_infer ./ch_PP-OCRv3_rec_infer ./ppocr_keys_v1.txt ./12.jpg 1 # 在CPU上使用ONNX Runtime推理 ./infer_demo ./ch_PP-OCRv3_det_infer ./ch_ppocr_mobile_v2.0_cls_infer ./ch_PP-OCRv3_rec_infer ./ppocr_keys_v1.txt ./12.jpg 2 @@ -110,7 +110,7 @@ wget https://gitee.com/paddlepaddle/PaddleOCR/raw/release/2.6/ppocr/utils/dict/l |数字选项|含义| |:---:|:---:| |0| 在CPU上使用Paddle Inference推理 | -|1| 在CPU上使用OenVINO推理 | +|1| 在CPU上使用OpenVINO推理 | |2| 在CPU上使用ONNX Runtime推理 | |3| 在CPU上使用Paddle Lite推理 | |4| 在GPU上使用Paddle Inference推理 | diff --git a/fastdeploy/core/float16.h b/fastdeploy/core/float16.h index 5b08b113d44..3afe195cf06 100644 --- a/fastdeploy/core/float16.h +++ b/fastdeploy/core/float16.h @@ -571,10 +571,10 @@ inline bool operator>=(const float16& a, const float16& b) { namespace std { -#if defined(__linux__) && !defined(__ANDROID__) && defined(_LIBCPP_VERSION) +#if defined(_LIBCPP_VERSION) -// TODO: 看看是怎么一回事 -// clang + libc++ 在 linux 下无法编译下面的代码 +// LLVM libc++ prohibits user specialization of standard type traits (marked with _LIBCPP_NO_SPECIALIZATIONS) +// and std::is_pod is deprecated in C++20. #else diff --git a/fastdeploy/runtime/backends/ort/option.h b/fastdeploy/runtime/backends/ort/option.h index 595a2fcfde9..bb10608acaa 100755 --- a/fastdeploy/runtime/backends/ort/option.h +++ b/fastdeploy/runtime/backends/ort/option.h @@ -21,6 +21,9 @@ #include #include #include + +struct OrtSessionOptions; + namespace fastdeploy { /*! @brief Option object to configure ONNX Runtime backend @@ -55,5 +58,8 @@ struct OrtBackendOption { void DisableOrtFP16OpTypes(const std::vector& ops) { ort_disabled_ops_.insert(ort_disabled_ops_.end(), ops.begin(), ops.end()); } + + bool (*configure_session_callback)(OrtSessionOptions* session_options, void* user_data) = nullptr; + void* configure_session_callback_data = nullptr; }; } // namespace fastdeploy diff --git a/fastdeploy/runtime/backends/ort/ort_backend.cc b/fastdeploy/runtime/backends/ort/ort_backend.cc index 9ef07497224..03c59a665eb 100644 --- a/fastdeploy/runtime/backends/ort/ort_backend.cc +++ b/fastdeploy/runtime/backends/ort/ort_backend.cc @@ -53,6 +53,9 @@ std::wstring ToWstring(const std::string& str) { bool OrtBackend::BuildOption(const OrtBackendOption& option) { option_ = option; + if (option_.configure_session_callback) { + return option_.configure_session_callback(static_cast(session_options_), option_.configure_session_callback_data); + } if (option.graph_optimization_level >= 0) { session_options_.SetGraphOptimizationLevel( GraphOptimizationLevel(option.graph_optimization_level)); @@ -313,7 +316,7 @@ bool OrtBackend::InitFromOnnx(const std::string& model_file, paddle2onnx::ConvertFP32ToFP16(model_file.c_str(), model_file.size(), &model_content_ptr, &model_content_size); #else - FDERROR << "Didn't compile with PaddlePaddle Frontend, FP16 is not supported" << std::endl; + FDERROR << "Didn't compile with ENABLE_PADDLE2ONNX, FP16 is not supported" << std::endl; return false; #endif std::string onnx_model_proto(model_content_ptr, diff --git a/fastdeploy/vision.h b/fastdeploy/vision.h index 3da49fa5f52..ef4736e7592 100755 --- a/fastdeploy/vision.h +++ b/fastdeploy/vision.h @@ -64,6 +64,8 @@ #include "fastdeploy/vision/ocr/ppocr/ppocr_v2.h" #include "fastdeploy/vision/ocr/ppocr/ppocr_v3.h" #include "fastdeploy/vision/ocr/ppocr/ppocr_v4.h" +#include "fastdeploy/vision/ocr/ppocr/ppocr_v5.h" +#include "fastdeploy/vision/ocr/ppocr/ppocr_v6.h" #include "fastdeploy/vision/ocr/ppocr/ppstructurev2_table.h" #include "fastdeploy/vision/ocr/ppocr/ppstructurev2_layout.h" #include "fastdeploy/vision/ocr/ppocr/recognizer.h" diff --git a/fastdeploy/vision/ocr/ocr_pybind.cc b/fastdeploy/vision/ocr/ocr_pybind.cc index 0005c724260..3e96fcb4a2c 100755 --- a/fastdeploy/vision/ocr/ocr_pybind.cc +++ b/fastdeploy/vision/ocr/ocr_pybind.cc @@ -17,6 +17,8 @@ namespace fastdeploy { void BindPPOCRModel(pybind11::module& m); +void BindPPOCRv6(pybind11::module& m); +void BindPPOCRv5(pybind11::module& m); void BindPPOCRv4(pybind11::module& m); void BindPPOCRv3(pybind11::module& m); void BindPPOCRv2(pybind11::module& m); @@ -25,6 +27,8 @@ void BindPPStructureV2Table(pybind11::module& m); void BindOcr(pybind11::module& m) { auto ocr_module = m.def_submodule("ocr", "Module to deploy OCR models"); BindPPOCRModel(ocr_module); + BindPPOCRv6(ocr_module); + BindPPOCRv5(ocr_module); BindPPOCRv4(ocr_module); BindPPOCRv3(ocr_module); BindPPOCRv2(ocr_module); diff --git a/fastdeploy/vision/ocr/ppocr/det_postprocessor.cc b/fastdeploy/vision/ocr/ppocr/det_postprocessor.cc index 428142fd2e8..edf69637a76 100644 --- a/fastdeploy/vision/ocr/ppocr/det_postprocessor.cc +++ b/fastdeploy/vision/ocr/ppocr/det_postprocessor.cc @@ -51,7 +51,7 @@ bool DBDetectorPostprocessor::SingleBatchPostprocessor( boxes = util_post_processor_.BoxesFromBitmap( pred_map, bit_map, det_db_box_thresh_, det_db_unclip_ratio_, - det_db_score_mode_); + det_db_score_mode_, max_candidates_); boxes = util_post_processor_.FilterTagDetRes(boxes, det_img_info); diff --git a/fastdeploy/vision/ocr/ppocr/det_postprocessor.h b/fastdeploy/vision/ocr/ppocr/det_postprocessor.h index fc0d8c84d2a..fda08537dec 100644 --- a/fastdeploy/vision/ocr/ppocr/det_postprocessor.h +++ b/fastdeploy/vision/ocr/ppocr/det_postprocessor.h @@ -67,6 +67,12 @@ class FASTDEPLOY_DECL DBDetectorPostprocessor { /// Get use_dilation of the detection postprocess int GetUseDilation() const { return use_dilation_; } + /// Set det_db_max_candidates for the detection postprocess, default is 1000 + void SetDetDBMaxCandidates(int max_candidates) { + max_candidates_ = max_candidates; + } + /// Get det_db_max_candidates of the detection postprocess + int GetDetDBMaxCandidates() const { return max_candidates_; } private: double det_db_thresh_ = 0.3; @@ -74,6 +80,7 @@ class FASTDEPLOY_DECL DBDetectorPostprocessor { double det_db_unclip_ratio_ = 1.5; std::string det_db_score_mode_ = "slow"; bool use_dilation_ = false; + int max_candidates_ = 1000; PostProcessor util_post_processor_; bool SingleBatchPostprocessor(const float* out_data, int n2, int n3, const std::array& det_img_info, diff --git a/fastdeploy/vision/ocr/ppocr/ocrmodel_pybind.cc b/fastdeploy/vision/ocr/ppocr/ocrmodel_pybind.cc index b468a20d2d7..4659e409be8 100644 --- a/fastdeploy/vision/ocr/ppocr/ocrmodel_pybind.cc +++ b/fastdeploy/vision/ocr/ppocr/ocrmodel_pybind.cc @@ -77,6 +77,9 @@ void BindPPOCRModel(pybind11::module& m) { .def_property("use_dilation", &vision::ocr::DBDetectorPostprocessor::GetUseDilation, &vision::ocr::DBDetectorPostprocessor::SetUseDilation) + .def_property("det_db_max_candidates", + &vision::ocr::DBDetectorPostprocessor::GetDetDBMaxCandidates, + &vision::ocr::DBDetectorPostprocessor::SetDetDBMaxCandidates) .def("run", [](vision::ocr::DBDetectorPostprocessor& self, diff --git a/fastdeploy/vision/ocr/ppocr/ppocr_pybind.cc b/fastdeploy/vision/ocr/ppocr/ppocr_pybind.cc index 91826e48b5b..8d440afcdd7 100755 --- a/fastdeploy/vision/ocr/ppocr/ppocr_pybind.cc +++ b/fastdeploy/vision/ocr/ppocr/ppocr_pybind.cc @@ -16,6 +16,72 @@ #include "fastdeploy/pybind/main.h" namespace fastdeploy { +void BindPPOCRv6(pybind11::module& m) { + // PPOCRv6 + pybind11::class_(m, "PPOCRv6") + + .def(pybind11::init()) + .def(pybind11::init()) + .def_property("cls_batch_size", &pipeline::PPOCRv6::GetClsBatchSize, + &pipeline::PPOCRv6::SetClsBatchSize) + .def_property("rec_batch_size", &pipeline::PPOCRv6::GetRecBatchSize, + &pipeline::PPOCRv6::SetRecBatchSize) + .def("clone", [](pipeline::PPOCRv6& self) { return self.Clone(); }) + .def("predict", + [](pipeline::PPOCRv6& self, pybind11::array& data) { + auto mat = PyArrayToCvMat(data); + vision::OCRResult res; + self.Predict(&mat, &res); + return res; + }) + .def("batch_predict", + [](pipeline::PPOCRv6& self, std::vector& data) { + std::vector images; + for (size_t i = 0; i < data.size(); ++i) { + images.push_back(PyArrayToCvMat(data[i])); + } + std::vector results; + self.BatchPredict(images, &results); + return results; + }); +} + +void BindPPOCRv5(pybind11::module& m) { + // PPOCRv5 + pybind11::class_(m, "PPOCRv5") + + .def(pybind11::init()) + .def(pybind11::init()) + .def_property("cls_batch_size", &pipeline::PPOCRv5::GetClsBatchSize, + &pipeline::PPOCRv5::SetClsBatchSize) + .def_property("rec_batch_size", &pipeline::PPOCRv5::GetRecBatchSize, + &pipeline::PPOCRv5::SetRecBatchSize) + .def("clone", [](pipeline::PPOCRv5& self) { return self.Clone(); }) + .def("predict", + [](pipeline::PPOCRv5& self, pybind11::array& data) { + auto mat = PyArrayToCvMat(data); + vision::OCRResult res; + self.Predict(&mat, &res); + return res; + }) + .def("batch_predict", + [](pipeline::PPOCRv5& self, std::vector& data) { + std::vector images; + for (size_t i = 0; i < data.size(); ++i) { + images.push_back(PyArrayToCvMat(data[i])); + } + std::vector results; + self.BatchPredict(images, &results); + return results; + }); +} + void BindPPOCRv4(pybind11::module& m) { // PPOCRv4 pybind11::class_(m, "PPOCRv4") diff --git a/fastdeploy/vision/ocr/ppocr/ppocr_v5.h b/fastdeploy/vision/ocr/ppocr/ppocr_v5.h new file mode 100644 index 00000000000..6c9ecca4b4b --- /dev/null +++ b/fastdeploy/vision/ocr/ppocr/ppocr_v5.h @@ -0,0 +1,90 @@ +// Copyright (c) 2023 PaddlePaddle Authors. All Rights Reserved. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +#pragma once + +#include "fastdeploy/vision/ocr/ppocr/ppocr_v4.h" + +namespace fastdeploy { +/** \brief This pipeline can launch detection model, classification model and recognition model sequentially. All OCR pipeline APIs are defined inside this namespace. + * + */ +namespace pipeline { +/*! @brief PPOCRv5 is used to load PP-OCRv5 series models provided by PaddleOCR. + */ +class FASTDEPLOY_DECL PPOCRv5 : public PPOCRv4 { + public: + /** \brief Set up the detection model path, classification model path and recognition model path respectively. + * + * \param[in] det_model Path of detection model, e.g ./ch_PP-OCRv5_det_infer + * \param[in] cls_model Path of classification model, e.g ./ch_ppocr_mobile_v2.0_cls_infer + * \param[in] rec_model Path of recognition model, e.g ./ch_PP-OCRv5_rec_infer + */ + PPOCRv5(fastdeploy::vision::ocr::DBDetector* det_model, + fastdeploy::vision::ocr::Classifier* cls_model, + fastdeploy::vision::ocr::Recognizer* rec_model) + : PPOCRv4(det_model, cls_model, rec_model) { + auto preprocess_shape = recognizer_->GetPreprocessor().GetRecImageShape(); + preprocess_shape[1] = 48; + recognizer_->GetPreprocessor().SetRecImageShape(preprocess_shape); + if (detector_ != nullptr) { + detector_->GetPostprocessor().SetDetDBThresh(0.3); + detector_->GetPostprocessor().SetDetDBBoxThresh(0.6); + detector_->GetPostprocessor().SetDetDBUnclipRatio(1.5); + detector_->GetPostprocessor().SetDetDBMaxCandidates(1000); + } + } + /** \brief Classification model is optional, so this function is set up the detection model path and recognition model path respectively. + * + * \param[in] det_model Path of detection model, e.g ./ch_PP-OCRv5_det_infer + * \param[in] rec_model Path of recognition model, e.g ./ch_PP-OCRv5_rec_infer + */ + PPOCRv5(fastdeploy::vision::ocr::DBDetector* det_model, + fastdeploy::vision::ocr::Recognizer* rec_model) + : PPOCRv4(det_model, rec_model) { + auto preprocess_shape = recognizer_->GetPreprocessor().GetRecImageShape(); + preprocess_shape[1] = 48; + recognizer_->GetPreprocessor().SetRecImageShape(preprocess_shape); + if (detector_ != nullptr) { + detector_->GetPostprocessor().SetDetDBThresh(0.3); + detector_->GetPostprocessor().SetDetDBBoxThresh(0.6); + detector_->GetPostprocessor().SetDetDBUnclipRatio(1.5); + detector_->GetPostprocessor().SetDetDBMaxCandidates(1000); + } + } + + /** \brief Clone a new PPOCRv5 with less memory usage when multiple instances of the same model are created + * + * \return new PPOCRv5* type unique pointer + */ + std::unique_ptr Clone() const { + std::unique_ptr clone_model = utils::make_unique(PPOCRv5(*this)); + clone_model->detector_ = detector_->Clone().release(); + if (classifier_ != nullptr) { + clone_model->classifier_ = classifier_->Clone().release(); + } + clone_model->recognizer_ = recognizer_->Clone().release(); + return clone_model; + } +}; + +} // namespace pipeline + +namespace application { +namespace ocrsystem { + typedef pipeline::PPOCRv5 PPOCRSystemv5; +} // namespace ocrsystem +} // namespace application + +} // namespace fastdeploy diff --git a/fastdeploy/vision/ocr/ppocr/ppocr_v6.h b/fastdeploy/vision/ocr/ppocr/ppocr_v6.h new file mode 100644 index 00000000000..d4ca7f6e3ee --- /dev/null +++ b/fastdeploy/vision/ocr/ppocr/ppocr_v6.h @@ -0,0 +1,90 @@ +// Copyright (c) 2024 PaddlePaddle Authors. All Rights Reserved. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +#pragma once + +#include "fastdeploy/vision/ocr/ppocr/ppocr_v5.h" + +namespace fastdeploy { +/** \brief This pipeline can launch detection model, classification model and recognition model sequentially. All OCR pipeline APIs are defined inside this namespace. + * + */ +namespace pipeline { +/*! @brief PPOCRv6 is used to load PP-OCRv6 series models provided by PaddleOCR. + */ +class FASTDEPLOY_DECL PPOCRv6 : public PPOCRv5 { + public: + /** \brief Set up the detection model path, classification model path and recognition model path respectively. + * + * \param[in] det_model Path of detection model, e.g ./ch_PP-OCRv6_det_infer + * \param[in] cls_model Path of classification model, e.g ./ch_ppocr_mobile_v2.0_cls_infer + * \param[in] rec_model Path of recognition model, e.g ./ch_PP-OCRv6_rec_infer + */ + PPOCRv6(fastdeploy::vision::ocr::DBDetector* det_model, + fastdeploy::vision::ocr::Classifier* cls_model, + fastdeploy::vision::ocr::Recognizer* rec_model) + : PPOCRv5(det_model, cls_model, rec_model) { + auto preprocess_shape = recognizer_->GetPreprocessor().GetRecImageShape(); + preprocess_shape[1] = 48; + recognizer_->GetPreprocessor().SetRecImageShape(preprocess_shape); + if (detector_ != nullptr) { + detector_->GetPostprocessor().SetDetDBThresh(0.2); + detector_->GetPostprocessor().SetDetDBBoxThresh(0.45); + detector_->GetPostprocessor().SetDetDBUnclipRatio(1.4); + detector_->GetPostprocessor().SetDetDBMaxCandidates(3000); + } + } + /** \brief Classification model is optional, so this function is set up the detection model path and recognition model path respectively. + * + * \param[in] det_model Path of detection model, e.g ./ch_PP-OCRv6_det_infer + * \param[in] rec_model Path of recognition model, e.g ./ch_PP-OCRv6_rec_infer + */ + PPOCRv6(fastdeploy::vision::ocr::DBDetector* det_model, + fastdeploy::vision::ocr::Recognizer* rec_model) + : PPOCRv5(det_model, rec_model) { + auto preprocess_shape = recognizer_->GetPreprocessor().GetRecImageShape(); + preprocess_shape[1] = 48; + recognizer_->GetPreprocessor().SetRecImageShape(preprocess_shape); + if (detector_ != nullptr) { + detector_->GetPostprocessor().SetDetDBThresh(0.2); + detector_->GetPostprocessor().SetDetDBBoxThresh(0.45); + detector_->GetPostprocessor().SetDetDBUnclipRatio(1.4); + detector_->GetPostprocessor().SetDetDBMaxCandidates(3000); + } + } + + /** \brief Clone a new PPOCRv6 with less memory usage when multiple instances of the same model are created + * + * \return new PPOCRv6* type unique pointer + */ + std::unique_ptr Clone() const { + std::unique_ptr clone_model = utils::make_unique(PPOCRv6(*this)); + clone_model->detector_ = detector_->Clone().release(); + if (classifier_ != nullptr) { + clone_model->classifier_ = classifier_->Clone().release(); + } + clone_model->recognizer_ = recognizer_->Clone().release(); + return clone_model; + } +}; + +} // namespace pipeline + +namespace application { +namespace ocrsystem { + typedef pipeline::PPOCRv6 PPOCRSystemv6; +} // namespace ocrsystem +} // namespace application + +} // namespace fastdeploy diff --git a/fastdeploy/vision/ocr/ppocr/utils/ocr_postprocess_op.cc b/fastdeploy/vision/ocr/ppocr/utils/ocr_postprocess_op.cc index 7a8f387e233..49b46860841 100755 --- a/fastdeploy/vision/ocr/ppocr/utils/ocr_postprocess_op.cc +++ b/fastdeploy/vision/ocr/ppocr/utils/ocr_postprocess_op.cc @@ -244,9 +244,9 @@ float PostProcessor::BoxScoreFast(std::vector> box_array, std::vector>> PostProcessor::BoxesFromBitmap( const cv::Mat pred, const cv::Mat bitmap, const float &box_thresh, - const float &det_db_unclip_ratio, const std::string &det_db_score_mode) { + const float &det_db_unclip_ratio, const std::string &det_db_score_mode, + const int &max_candidates) { const int min_size = 3; - const int max_candidates = 1000; int width = bitmap.cols; int height = bitmap.rows; diff --git a/fastdeploy/vision/ocr/ppocr/utils/ocr_postprocess_op.h b/fastdeploy/vision/ocr/ppocr/utils/ocr_postprocess_op.h index 778f618edb2..9a57cec95bb 100644 --- a/fastdeploy/vision/ocr/ppocr/utils/ocr_postprocess_op.h +++ b/fastdeploy/vision/ocr/ppocr/utils/ocr_postprocess_op.h @@ -53,7 +53,8 @@ class PostProcessor { std::vector>> BoxesFromBitmap( const cv::Mat pred, const cv::Mat bitmap, const float &box_thresh, - const float &det_db_unclip_ratio, const std::string &det_db_score_mode); + const float &det_db_unclip_ratio, const std::string &det_db_score_mode, + const int &max_candidates = 1000); std::vector>> FilterTagDetRes( std::vector>> boxes, diff --git a/fastdeploy/vision/utils/cosine_similarity.cc b/fastdeploy/vision/utils/cosine_similarity.cc index 20482ada99c..881e79ce701 100644 --- a/fastdeploy/vision/utils/cosine_similarity.cc +++ b/fastdeploy/vision/utils/cosine_similarity.cc @@ -23,24 +23,19 @@ float CosineSimilarity(const std::vector& a, const std::vector& b, FDASSERT((a.size() == b.size()) && (a.size() != 0), "The size of a and b must be equal and >= 1."); size_t num_val = a.size(); + float mul_ab = 0.f; if (normalized) { - float mul_a = 0.f, mul_b = 0.f, mul_ab = 0.f; for (size_t i = 0; i < num_val; ++i) { - mul_a += (a[i] * a[i]); - mul_b += (b[i] * b[i]); mul_ab += (a[i] * b[i]); } - return (mul_ab / (std::sqrt(mul_a) * std::sqrt(mul_b))); + return mul_ab; } auto norm_a = L2Normalize(a); auto norm_b = L2Normalize(b); - float mul_a = 0.f, mul_b = 0.f, mul_ab = 0.f; for (size_t i = 0; i < num_val; ++i) { - mul_a += (norm_a[i] * norm_a[i]); - mul_b += (norm_b[i] * norm_b[i]); mul_ab += (norm_a[i] * norm_b[i]); } - return (mul_ab / (std::sqrt(mul_a) * std::sqrt(mul_b))); + return mul_ab; } } // namespace utils diff --git a/paddle2onnx/CMakeLists.txt b/paddle2onnx/CMakeLists.txt deleted file mode 100644 index e69de29bb2d..00000000000 diff --git a/paddle2onnx/__init__.py b/paddle2onnx/__init__.py deleted file mode 100755 index 7d726e8b743..00000000000 --- a/paddle2onnx/__init__.py +++ /dev/null @@ -1,65 +0,0 @@ -# Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from paddle2onnx.utils import logging -from . import command -from .convert import dygraph2onnx -from .convert import program2onnx -from .version import version -from .version import git_version - -__version__ = version -__commit_id__ = git_version - - -def run_convert(model, input_shape_dict=None, scope=None, opset_version=9): - logging.warning( - "[Deprecated] `paddle2onnx.run_convert` will be deprecated in the future version, the recommended usage is `paddle2onnx.export`" - ) - from paddle2onnx.legacy import run_convert - return run_convert(model, input_shape_dict, scope, opset_version) - - -def export(model_file, - params_file="", - save_file=None, - opset_version=11, - auto_upgrade_opset=True, - verbose=True, - enable_onnx_checker=True, - enable_experimental_op=True, - enable_optimize=True, - custom_op_info=None, - deploy_backend="onnxruntime", - calibration_file="", - external_file="", - export_fp16_model=False): - import paddle2onnx.paddle2onnx_cpp2py_export as c_p2o - deploy_backend = deploy_backend.lower() - if custom_op_info is None: - onnx_model_str = c_p2o.export( - model_file, params_file, opset_version, auto_upgrade_opset, verbose, - enable_onnx_checker, enable_experimental_op, enable_optimize, {}, - deploy_backend, calibration_file, external_file, export_fp16_model) - else: - onnx_model_str = c_p2o.export( - model_file, params_file, opset_version, auto_upgrade_opset, verbose, - enable_onnx_checker, enable_experimental_op, enable_optimize, - custom_op_info, deploy_backend, calibration_file, external_file, - export_fp16_model) - if save_file is not None: - with open(save_file, "wb") as f: - f.write(onnx_model_str) - else: - return onnx_model_str diff --git a/paddle2onnx/command.py b/paddle2onnx/command.py deleted file mode 100755 index 4562a9a6f48..00000000000 --- a/paddle2onnx/command.py +++ /dev/null @@ -1,284 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import -from six import text_type as _text_type -import argparse -import ast -import sys -import os -from paddle2onnx.utils import logging - - -def str2list(v): - if len(v) == 0: - return None - v = v.replace(" ", "") - v = eval(v) - return v - - -def arg_parser(): - parser = argparse.ArgumentParser() - parser.add_argument( - "--model_dir", - "-m", - type=_text_type, - default=None, - help="PaddlePaddle model directory, if params stored in single file, you need define '--model_filename' and 'params_filename'." - ) - parser.add_argument( - "--model_filename", - "-mf", - type=_text_type, - default=None, - help="PaddlePaddle model's network file name, which under directory seted by --model_dir" - ) - parser.add_argument( - "--params_filename", - "-pf", - type=_text_type, - default=None, - help="PaddlePaddle model's param file name(param files combined in single file), which under directory seted by --model_dir." - ) - parser.add_argument( - "--save_file", - "-s", - type=_text_type, - default=None, - help="file path to save onnx model") - parser.add_argument( - "--opset_version", - "-ov", - type=int, - default=9, - help="set onnx opset version to export") - parser.add_argument( - "--input_shape_dict", - "-isd", - type=_text_type, - default="None", - help="define input shapes, e.g --input_shape_dict=\"{'image':[1, 3, 608, 608]}\" or" \ - "--input_shape_dict=\"{'image':[1, 3, 608, 608], 'im_shape': [1, 2], 'scale_factor': [1, 2]}\"") - parser.add_argument( - "--enable_dev_version", - type=ast.literal_eval, - default=True, - help="whether to use new version of Paddle2ONNX which is under developing, default True" - ) - parser.add_argument( - "--deploy_backend", - "-d", - type=_text_type, - default="onnxruntime", - choices=["onnxruntime", "tensorrt", "rknn", "others"], - help="Quantize model deploy backend, default onnxruntime.") - parser.add_argument( - "--save_calibration_file", - type=_text_type, - default="calibration.cache", - help="The calibration cache for TensorRT deploy, default calibration.cache." - ) - parser.add_argument( - "--enable_onnx_checker", - type=ast.literal_eval, - default=True, - help="whether check onnx model validity, default True") - parser.add_argument( - "--enable_paddle_fallback", - type=ast.literal_eval, - default=False, - help="whether use PaddleFallback for custom op, default is False") - parser.add_argument( - "--version", - "-v", - action="store_true", - default=False, - help="get version of paddle2onnx") - parser.add_argument( - "--output_names", - "-on", - type=str2list, - default=None, - help="define output names, e.g --output_names=\"[\"output1\"]\" or \ - --output_names=\"[\"output1\", \"output2\", \"output3\"]\" or \ - --output_names=\"{\"Paddleoutput\":\"Onnxoutput\"}\"") - parser.add_argument( - "--enable_auto_update_opset", - type=ast.literal_eval, - default=True, - help="whether enable auto_update_opset, default is True") - parser.add_argument( - "--external_filename", - type=_text_type, - default=None, - help="The filename of external_data when the model is bigger than 2G.") - parser.add_argument( - "--export_fp16_model", - type=ast.literal_eval, - default=False, - help="Whether export FP16 model for ORT-GPU, default False") - parser.add_argument( - "--custom_ops", - type=_text_type, - default="{}", - help="Ops that needs to be converted to custom op, e.g --custom_ops '{\"paddle_op\":\"onnx_op\"}', default {}" - ) - return parser - - -def c_paddle_to_onnx(model_file, - params_file="", - save_file=None, - opset_version=7, - auto_upgrade_opset=True, - verbose=True, - enable_onnx_checker=True, - enable_experimental_op=True, - enable_optimize=True, - deploy_backend="onnxruntime", - calibration_file="", - external_file="", - export_fp16_model=False, - custom_ops={}): - import paddle2onnx.paddle2onnx_cpp2py_export as c_p2o - onnx_model_str = c_p2o.export( - model_file, params_file, opset_version, auto_upgrade_opset, verbose, - enable_onnx_checker, enable_experimental_op, enable_optimize, - custom_ops, deploy_backend, calibration_file, external_file, - export_fp16_model) - if save_file is not None: - with open(save_file, "wb") as f: - f.write(onnx_model_str) - else: - return onnx_model_str - - -def program2onnx(model_dir, - save_file, - model_filename=None, - params_filename=None, - opset_version=9, - enable_onnx_checker=False, - operator_export_type="ONNX", - input_shape_dict=None, - output_names=None, - auto_update_opset=True): - logging.warning( - "[Deprecated] `paddle2onnx.command.program2onnx` will be deprecated in the future version, the recommended usage is `paddle2onnx.export`" - ) - from paddle2onnx.legacy.command import program2onnx - return program2onnx(model_dir, save_file, model_filename, params_filename, - opset_version, enable_onnx_checker, - operator_export_type, input_shape_dict, output_names, - auto_update_opset) - - -def main(): - if len(sys.argv) < 2: - logging.info("Use \"paddle2onnx -h\" to print the help information") - logging.info( - "For more information, please follow our github repo below:") - logging.info("Github: https://github.com/PaddlePaddle/paddle2onnx.git") - return - - parser = arg_parser() - args = parser.parse_args() - - if args.version: - import paddle2onnx - logging.info("paddle2onnx-{} with python>=3.6, paddlepaddle>=2.0.0". - format(paddle2onnx.__version__)) - return - - assert args.model_dir is not None, "--model_dir should be defined while translating paddle model to onnx" - assert args.save_file is not None, "--save_file should be defined while translating paddle model to onnx" - - input_shape_dict = eval(args.input_shape_dict) - - operator_export_type = "ONNX" - if args.enable_paddle_fallback: - logging.warning( - "[Deprecated] The flag `--enable_paddle_fallback` will be deprecated, and only works while `--enable_dev_version False` now." - ) - operator_export_type = "PaddleFallback" - - if args.output_names is not None and args.enable_dev_version: - logging.warning( - "[Deprecated] The flag `--output_names` is deprecated, if you need to modify the output name, please refer to this tool https://github.com/jiangjiajun/PaddleUtils/tree/main/onnx " - ) - if not isinstance(args.output_names, (list, dict)): - raise TypeError( - "The output_names should be 'list' or 'dict', but received type is %s." - % type(args.output_names)) - - if input_shape_dict is not None and args.enable_dev_version: - logging.warning( - "[Deprecated] The flag `--input_shape_dict` is deprecated, if you need to modify the input shape of PaddlePaddle model, please refer to this tool https://github.com/jiangjiajun/PaddleUtils/tree/main/paddle " - ) - - if args.enable_dev_version: - model_file = os.path.join(args.model_dir, args.model_filename) - if args.params_filename is None: - params_file = "" - else: - params_file = os.path.join(args.model_dir, args.params_filename) - - if args.external_filename is None: - args.external_filename = "external_data" - - base_path = os.path.dirname(args.save_file) - if base_path and not os.path.exists(base_path): - os.mkdir(base_path) - external_file = os.path.join(base_path, args.external_filename) - - custom_ops_dict = eval(args.custom_ops) - - calibration_file = args.save_calibration_file - c_paddle_to_onnx( - model_file=model_file, - params_file=params_file, - save_file=args.save_file, - opset_version=args.opset_version, - auto_upgrade_opset=args.enable_auto_update_opset, - verbose=True, - enable_onnx_checker=args.enable_onnx_checker, - enable_experimental_op=True, - enable_optimize=True, - deploy_backend=args.deploy_backend, - calibration_file=calibration_file, - external_file=external_file, - export_fp16_model=args.export_fp16_model, - custom_ops=custom_ops_dict) - logging.info("===============Make PaddlePaddle Better!================") - logging.info("A little survey: https://iwenjuan.baidu.com/?code=r8hu2s") - return - - program2onnx( - args.model_dir, - args.save_file, - args.model_filename, - args.params_filename, - opset_version=args.opset_version, - enable_onnx_checker=args.enable_onnx_checker, - operator_export_type=operator_export_type, - input_shape_dict=input_shape_dict, - output_names=args.output_names, - auto_update_opset=args.enable_auto_update_opset) - logging.info("===============Make PaddlePaddle Better!================") - logging.info("A little survey: https://iwenjuan.baidu.com/?code=r8hu2s") - - -if __name__ == "__main__": - main() diff --git a/paddle2onnx/convert.py b/paddle2onnx/convert.py deleted file mode 100755 index 50c8db2da36..00000000000 --- a/paddle2onnx/convert.py +++ /dev/null @@ -1,83 +0,0 @@ -# Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from paddle2onnx.utils import logging - - -def export_onnx(paddle_graph, - save_file, - opset_version=9, - enable_onnx_checker=False, - operator_export_type="ONNX", - verbose=False, - auto_update_opset=True, - output_names=None): - from paddle2onnx.legacy.convert import export_onnx - return export_onnx(paddle_graph, save_file, opset_version, opset_version, - enable_onnx_checker, operator_export_type, verbose, - auto_update_opset, output_names) - - -def dygraph2onnx(layer, save_file, input_spec=None, opset_version=9, **configs): - if "enable_dev_version" in configs and not configs["enable_dev_version"]: - from paddle2onnx.legacy.convert import dygraph2onnx - return dygraph2onnx(layer, save_file, input_spec, opset_version, - **configs) - - import os - import paddle2onnx - import paddle - dirname = os.path.split(save_file)[0] - paddle_model_dir = os.path.join(dirname, - "paddle_model_static_onnx_temp_dir") - model_file = os.path.join(paddle_model_dir, "model.pdmodel") - params_file = os.path.join(paddle_model_dir, "model.pdiparams") - - if os.path.exists(paddle_model_dir): - if os.path.isfile(paddle_model_dir): - logging.info("File {} exists, will remove it.".format( - paddle_model_dir)) - os.remove(paddle_model_dir) - if os.path.isfile(model_file): - os.remove(model_file) - if os.path.isfile(params_file): - os.remove(params_file) - paddle.jit.save(layer, os.path.join(paddle_model_dir, "model"), input_spec) - logging.info("Static PaddlePaddle model saved in {}.".format( - paddle_model_dir)) - if not os.path.isfile(params_file): - params_file = "" - - if save_file is None: - return paddle2onnx.export(model_file, params_file, save_file, - opset_version) - else: - paddle2onnx.export(model_file, params_file, save_file, opset_version) - logging.info("ONNX model saved in {}.".format(save_file)) - - -def program2onnx(program, - scope, - save_file, - feed_var_names=None, - target_vars=None, - opset_version=9, - enable_onnx_checker=False, - operator_export_type="ONNX", - auto_update_opset=True, - **configs): - from paddle2onnx.legacy.convert import program2onnx - return program2onnx(program, scope, save_file, feed_var_names, target_vars, - opset_version, enable_onnx_checker, - operator_export_type, auto_update_opset, **configs) diff --git a/paddle2onnx/convert_to_fp16.py b/paddle2onnx/convert_to_fp16.py deleted file mode 100755 index b0321f5ae15..00000000000 --- a/paddle2onnx/convert_to_fp16.py +++ /dev/null @@ -1,38 +0,0 @@ -# Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -from __future__ import absolute_import - -import argparse -import sys -from paddle2onnx.utils import logging - - -def parse_arguments(): - parser = argparse.ArgumentParser() - parser.add_argument( - '--input_model_path', - required=True, - help='The path of input onnx model file.') - parser.add_argument( - '--output_model_path', - required=True, - help='The file path to write optimized onnx model file.') - return parser.parse_args() - - -if __name__ == '__main__': - args = parse_arguments() - import paddle2onnx.paddle2onnx_cpp2py_export as c_p2o - c_p2o.convert_to_fp16(args.input_model_path, args.output_model_path) - logging.info("FP16 model saved in {}.".format(args.output_model_path)) diff --git a/paddle2onnx/converter.cc b/paddle2onnx/converter.cc deleted file mode 100644 index 40cc63a5889..00000000000 --- a/paddle2onnx/converter.cc +++ /dev/null @@ -1,306 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/converter.h" - -#include -#include -#include -#include - -#include "paddle2onnx/mapper/exporter.h" -#include "paddle2onnx/optimizer/convert_fp32_to_fp16.h" - -namespace paddle2onnx { - -PADDLE2ONNX_DECL bool IsExportable(const char* model, const char* params, - int32_t opset_version, - bool auto_upgrade_opset, bool verbose, - bool enable_onnx_checker, - bool enable_experimental_op, - bool enable_optimize, CustomOp* ops, - int op_count, const char* deploy_backend) { - auto parser = PaddleParser(); - if (!parser.Init(model, params)) { - return false; - } - paddle2onnx::ModelExporter me; - std::set unsupported_ops; - if (!me.CheckIfOpSupported(parser, &unsupported_ops, - enable_experimental_op)) { - return false; - } - - // Add custom operator information - if (ops != nullptr && op_count > 0) { - for (int i = 0; i < op_count; ++i) { - std::string op_name(ops[i].op_name, strlen(ops[i].op_name)); - std::string export_op_name(ops[i].export_op_name, - strlen(ops[i].export_op_name)); - if (export_op_name == "paddle2onnx_null") { - export_op_name = op_name; - } - me.custom_ops[op_name] = export_op_name; - } - } - - if (me.GetMinOpset(parser, false) < 0) { - return false; - } - std::string calibration_str; - std::string onnx_model = - me.Run(parser, opset_version, auto_upgrade_opset, verbose, - enable_onnx_checker, enable_experimental_op, enable_optimize, - deploy_backend, &calibration_str); - if (onnx_model.empty()) { - P2OLogger(verbose) << "The exported ONNX model is invalid!" << std::endl; - return false; - } - if (parser.is_quantized_model && "tensorrt" == std::string(deploy_backend) && - calibration_str.empty()) { - P2OLogger(verbose) << "Can not generate calibration cache for TensorRT " - "deploy backend when export quantize model." - << std::endl; - return false; - } - return true; -} - -PADDLE2ONNX_DECL bool IsExportable(const void* model_buffer, int model_size, - const void* params_buffer, int params_size, - int32_t opset_version, - bool auto_upgrade_opset, bool verbose, - bool enable_onnx_checker, - bool enable_experimental_op, - bool enable_optimize, CustomOp* ops, - int op_count, const char* deploy_backend) { - auto parser = PaddleParser(); - if (!parser.Init(model_buffer, model_size, params_buffer, params_size)) { - return false; - } - paddle2onnx::ModelExporter me; - std::set unsupported_ops; - if (!me.CheckIfOpSupported(parser, &unsupported_ops, - enable_experimental_op)) { - return false; - } - - // Add custom operator information - if (ops != nullptr && op_count > 0) { - for (int i = 0; i < op_count; ++i) { - std::string op_name(ops[i].op_name, strlen(ops[i].op_name)); - std::string export_op_name(ops[i].export_op_name, - strlen(ops[i].export_op_name)); - if (export_op_name == "paddle2onnx_null") { - export_op_name = op_name; - } - me.custom_ops[op_name] = export_op_name; - } - } - - if (me.GetMinOpset(parser, false) < 0) { - return false; - } - std::string calibration_str; - std::string onnx_model = - me.Run(parser, opset_version, auto_upgrade_opset, verbose, - enable_onnx_checker, enable_experimental_op, enable_optimize, - deploy_backend, &calibration_str); - if (onnx_model.empty()) { - P2OLogger(verbose) << "The exported ONNX model is invalid!" << std::endl; - return false; - } - if (parser.is_quantized_model && "tensorrt" == std::string(deploy_backend) && - calibration_str.empty()) { - P2OLogger(verbose) << "Can not generate calibration cache for TensorRT " - "deploy backend when export quantize model." - << std::endl; - return false; - } - return true; -} - -PADDLE2ONNX_DECL bool Export( - const char* model, const char* params, char** out, int* out_size, - int32_t opset_version, bool auto_upgrade_opset, bool verbose, - bool enable_onnx_checker, bool enable_experimental_op, bool enable_optimize, - CustomOp* ops, int op_count, const char* deploy_backend, - char** calibration_cache, int* calibration_size, const char* external_file, - bool* save_external, bool export_fp16_model, char** disable_fp16_op_types, - int disable_fp16_op_types_count) { - auto parser = PaddleParser(); - P2OLogger(verbose) << "Start to parsing Paddle model..." << std::endl; - if (!parser.Init(model, params)) { - P2OLogger(verbose) << "Paddle model parsing failed." << std::endl; - return false; - } - paddle2onnx::ModelExporter me; - - // Add custom operator information - if (ops != nullptr && op_count > 0) { - for (int i = 0; i < op_count; ++i) { - std::string op_name(ops[i].op_name, strlen(ops[i].op_name)); - std::string export_op_name(ops[i].export_op_name, - strlen(ops[i].export_op_name)); - if (export_op_name == "paddle2onnx_null") { - export_op_name = op_name; - } - me.custom_ops[op_name] = export_op_name; - } - } - // Add disabled fp16 op information - std::vector disable_op_types; - if (disable_fp16_op_types != nullptr && disable_fp16_op_types_count > 0) { - for (int i = 0; i < disable_fp16_op_types_count; ++i) { - std::string disable_op_type(disable_fp16_op_types[i], - strlen(disable_fp16_op_types[i])); - disable_op_types.push_back(disable_op_type); - } - } - std::string calibration_str; - std::string result = me.Run( - parser, opset_version, auto_upgrade_opset, verbose, enable_onnx_checker, - enable_experimental_op, enable_optimize, deploy_backend, &calibration_str, - external_file, save_external, export_fp16_model, disable_op_types); - if (result.empty()) { - P2OLogger(verbose) << "The exported ONNX model is invalid!" << std::endl; - return false; - } - if (parser.is_quantized_model && "tensorrt" == std::string(deploy_backend) && - calibration_str.empty()) { - P2OLogger(verbose) << "Can not generate calibration cache for TensorRT " - "deploy backend when export quantize model." - << std::endl; - return false; - } - *out_size = result.size(); - *out = new char[*out_size](); - memcpy(*out, result.data(), *out_size); - if (calibration_str.size()) { - *calibration_size = calibration_str.size(); - *calibration_cache = new char[*calibration_size](); - memcpy(*calibration_cache, calibration_str.data(), *calibration_size); - } - return true; -} - -PADDLE2ONNX_DECL bool Export( - const void* model_buffer, int64_t model_size, const void* params_buffer, - int64_t params_size, char** out, int* out_size, int32_t opset_version, - bool auto_upgrade_opset, bool verbose, bool enable_onnx_checker, - bool enable_experimental_op, bool enable_optimize, CustomOp* ops, - int op_count, const char* deploy_backend, char** calibration_cache, - int* calibration_size, const char* external_file, bool* save_external, - bool export_fp16_model, char** disable_fp16_op_types, - int disable_fp16_op_types_count) { - auto parser = PaddleParser(); - P2OLogger(verbose) << "Start to parsing Paddle model..." << std::endl; - if (!parser.Init(model_buffer, model_size, params_buffer, params_size)) { - P2OLogger(verbose) << "Paddle model parsing failed." << std::endl; - return false; - } - paddle2onnx::ModelExporter me; - - // Add custom operator information - if (ops != nullptr && op_count > 0) { - for (int i = 0; i < op_count; ++i) { - std::string op_name(ops[i].op_name, strlen(ops[i].op_name)); - std::string export_op_name(ops[i].export_op_name, - strlen(ops[i].export_op_name)); - if (export_op_name == "paddle2onnx_null") { - export_op_name = op_name; - } - me.custom_ops[op_name] = export_op_name; - } - } - // Add disabled fp16 op information - std::vector disable_op_types; - if (disable_fp16_op_types != nullptr && disable_fp16_op_types_count > 0) { - for (int i = 0; i < disable_fp16_op_types_count; ++i) { - std::string disable_op_type(disable_fp16_op_types[i], - strlen(disable_fp16_op_types[i])); - disable_op_types.push_back(disable_op_type); - } - } - std::string calibration_str; - std::string result = me.Run( - parser, opset_version, auto_upgrade_opset, verbose, enable_onnx_checker, - enable_experimental_op, enable_optimize, deploy_backend, &calibration_str, - external_file, save_external, export_fp16_model, disable_op_types); - if (result.empty()) { - P2OLogger(verbose) << "The exported ONNX model is invalid!" << std::endl; - return false; - } - - if (parser.is_quantized_model && "tensorrt" == std::string(deploy_backend) && - calibration_str.empty()) { - P2OLogger(verbose) << "Can not generate calibration cache for TensorRT " - "deploy backend when export quantize model." - << std::endl; - return false; - } - *out_size = result.size(); - *out = new char[*out_size](); - memcpy(*out, result.data(), *out_size); - if (calibration_str.size()) { - *calibration_size = calibration_str.size(); - *calibration_cache = new char[*calibration_size](); - memcpy(*calibration_cache, calibration_str.data(), *calibration_size); - } - return true; -} - -PADDLE2ONNX_DECL bool ConvertFP32ToFP16(const char* onnx_model, int model_size, - char** out_model, int* out_model_size) { - std::string onnx_proto(onnx_model, onnx_model + model_size); - ONNX_NAMESPACE::ModelProto model; - model.ParseFromString(onnx_proto); - - P2OLogger(true) << "Convert FP32 ONNX model to FP16." << std::endl; - ConvertFp32ToFp16 convert; - convert.Convert(&model); - // save external data file for big model - std::string external_data_file; - if (model.ByteSizeLong() > INT_MAX) { - external_data_file = "external_data"; - } - paddle2onnx::ModelExporter me; - if (external_data_file.size()) { - me.SaveExternalData(model.mutable_graph(), external_data_file); - } - // check model - me.ONNXChecker(model, true); - - std::string result; - if (!model.SerializeToString(&result)) { - P2OLogger(true) - << "Error happenedd while optimizing the exported ONNX model." - << std::endl; - return false; - } - - *out_model_size = result.size(); - *out_model = new char[*out_model_size](); - memcpy(*out_model, result.data(), *out_model_size); - return true; -} - -ModelTensorInfo::~ModelTensorInfo() { - if (shape != nullptr) { - delete[] shape; - shape = nullptr; - rank = 0; - } -} -} // namespace paddle2onnx diff --git a/paddle2onnx/converter.h b/paddle2onnx/converter.h deleted file mode 100644 index 1be3f2b6dd0..00000000000 --- a/paddle2onnx/converter.h +++ /dev/null @@ -1,132 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. -#pragma once -#include - -#if defined(_WIN32) -#ifdef PADDLE2ONNX_LIB -#define PADDLE2ONNX_DECL __declspec(dllexport) -#else -#define PADDLE2ONNX_DECL __declspec(dllimport) -#endif // PADDLE2ONNX_LIB -#else -#define PADDLE2ONNX_DECL __attribute__((visibility("default"))) -#endif // _WIN32 - -namespace paddle2onnx { - -struct PADDLE2ONNX_DECL CustomOp { - char op_name[100] = "null"; - // if export_op_name set as "paddle2onnx_null" - // it will automaticly change to `op_name` - char export_op_name[100] = "paddle2onnx_null"; -}; - -PADDLE2ONNX_DECL bool IsExportable( - const char* model, const char* params, int32_t opset_version = 11, - bool auto_upgrade_opset = true, bool verbose = false, - bool enable_onnx_checker = true, bool enable_experimental_op = false, - bool enable_optimize = true, CustomOp* ops = nullptr, int op_count = 0, - const char* deploy_backend = "onnxruntime"); - -PADDLE2ONNX_DECL bool IsExportable( - const void* model_buffer, int model_size, const void* params_buffer, - int params_size, int32_t opset_version = 11, bool auto_upgrade_opset = true, - bool verbose = false, bool enable_onnx_checker = true, - bool enable_experimental_op = false, bool enable_optimize = true, - CustomOp* ops = nullptr, int op_count = 0, - const char* deploy_backend = "onnxruntime"); - -PADDLE2ONNX_DECL bool Export( - const char* model, const char* params, char** out, int* out_size, - int32_t opset_version = 11, bool auto_upgrade_opset = true, - bool verbose = false, bool enable_onnx_checker = true, - bool enable_experimental_op = false, bool enable_optimize = true, - CustomOp* ops = nullptr, int op_count = 0, - const char* deploy_backend = "onnxruntime", - char** calibration_cache = nullptr, int* calibration_size = 0, - const char* external_file = "", bool* save_external = nullptr, - bool export_fp16_model = false, char** disable_fp16_op_types = nullptr, - int disable_fp16_op_types_count = 0); - -PADDLE2ONNX_DECL bool Export( - const void* model_buffer, int64_t model_size, const void* params_buffer, - int64_t params_size, char** out, int* out_size, int32_t opset_version = 11, - bool auto_upgrade_opset = true, bool verbose = false, - bool enable_onnx_checker = true, bool enable_experimental_op = false, - bool enable_optimize = true, CustomOp* ops = nullptr, int op_count = 0, - const char* deploy_backend = "onnxruntime", - char** calibration_cache = nullptr, int* calibration_size = 0, - const char* external_file = "", bool* save_external = nullptr, - bool export_fp16_model = false, char** disable_fp16_op_types = nullptr, - int disable_fp16_op_types_count = 0); - -// Following are inside usage, will remove it maybe -struct PADDLE2ONNX_DECL ModelTensorInfo { - char name[100] = ""; - int64_t* shape = nullptr; - int32_t rank = 0; - // 0: float32 - // 1: double - // 2: uint8 - // 3: int8 - // 4: int32 - // 5: int64 - // 6: float16 - int32_t dtype = 0; - ~ModelTensorInfo(); -}; - -struct PADDLE2ONNX_DECL NMSParameters { - int64_t background_label = -1; - int64_t keep_top_k = 300; - float nms_eta = 1.0; - float nms_threshold = 0.7; - float score_threshold = 0.01; - int64_t nms_top_k = 10000; - bool normalized = true; -}; - -struct PADDLE2ONNX_DECL OnnxReader { - OnnxReader(const char* model_buffer, int buffer_size); - // suppose the maximum number of inputs/outputs is 100 - // suppose the longest string of inputs/outputs is 200 - // suppose the biggest rank will be less than 10 - ModelTensorInfo inputs[100]; - ModelTensorInfo outputs[100]; - int num_inputs; - int num_outputs; -}; - -PADDLE2ONNX_DECL bool RemoveMultiClassNMS(const char* onnx_model, - int model_size, char** out_model, - int* out_model_size); - -PADDLE2ONNX_DECL bool ConvertFP32ToFP16(const char* onnx_model, int model_size, - char** out_model, int* out_model_size); - -struct PADDLE2ONNX_DECL PaddleReader { - PaddleReader(const char* model_buffer, int buffer_size); - // suppose the maximum number of inputs/outputs is 100 - // suppose the longest string of inputs/outputs is 200 - ModelTensorInfo inputs[100]; - ModelTensorInfo outputs[100]; - int num_inputs; - int num_outputs; - bool has_nms = false; - bool is_quantize_model = false; - NMSParameters nms_params; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/cpp2py_export.cc b/paddle2onnx/cpp2py_export.cc deleted file mode 100644 index 0537d3c567d..00000000000 --- a/paddle2onnx/cpp2py_export.cc +++ /dev/null @@ -1,128 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include -#include - -#include -#include - -#include "paddle2onnx/converter.h" -#include "paddle2onnx/mapper/exporter.h" -#include "paddle2onnx/optimizer/paddle2onnx_optimizer.h" - -namespace paddle2onnx { - -typedef std::map CustomOpInfo; -PYBIND11_MODULE(paddle2onnx_cpp2py_export, m) { - m.doc() = "Paddle2ONNX: export PaddlePaddle to ONNX"; - m.def("export", [](const std::string& model_filename, - const std::string& params_filename, int opset_version = 9, - bool auto_upgrade_opset = true, bool verbose = true, - bool enable_onnx_checker = true, - bool enable_experimental_op = true, - bool enable_optimize = true, - const CustomOpInfo& info = CustomOpInfo(), - const std::string& deploy_backend = "onnxruntime", - const std::string& calibration_file = "", - const std::string& external_file = "", - const bool& export_fp16_model = false) { - P2OLogger(verbose) << "Start to parse PaddlePaddle model..." << std::endl; - P2OLogger(verbose) << "Model file path: " << model_filename << std::endl; - P2OLogger(verbose) << "Paramters file path: " << params_filename - << std::endl; - if (info.size() == 0) { - char* out = nullptr; - int size = 0; - char* calibration_cache = nullptr; - int cache_size = 0; - bool save_external; - if (!Export(model_filename.c_str(), params_filename.c_str(), &out, &size, - opset_version, auto_upgrade_opset, verbose, - enable_onnx_checker, enable_experimental_op, enable_optimize, - nullptr, 0, deploy_backend.c_str(), &calibration_cache, - &cache_size, external_file.c_str(), &save_external, - export_fp16_model)) { - P2OLogger(verbose) << "Paddle model convert failed." << std::endl; - return pybind11::bytes(""); - } - if (cache_size) { - std::string calibration_cache_str(calibration_cache, - calibration_cache + cache_size); - std::ofstream cache_file; - cache_file.open(calibration_file, std::ios::out); - cache_file << calibration_cache_str; - delete calibration_cache; - calibration_cache = nullptr; - P2OLogger(verbose) << "TensorRT calibration cache path: " - << calibration_file << std::endl; - } - std::string onnx_proto(out, out + size); - delete out; - out = nullptr; - return pybind11::bytes(onnx_proto); - } - - std::vector ops; - ops.resize(info.size()); - int index = 0; - for (auto& item : info) { - strcpy(ops[index].op_name, item.first.c_str()); - strcpy(ops[index].export_op_name, item.second.c_str()); - index += 1; - } - char* out = nullptr; - int size = 0; - char* calibration_cache = nullptr; - int cache_size = 0; - bool save_external; - if (!Export(model_filename.c_str(), params_filename.c_str(), &out, &size, - opset_version, auto_upgrade_opset, verbose, enable_onnx_checker, - enable_experimental_op, enable_optimize, ops.data(), - info.size(), deploy_backend.c_str(), &calibration_cache, - &cache_size, external_file.c_str(), &save_external, - export_fp16_model)) { - P2OLogger(verbose) << "Paddle model convert failed." << std::endl; - return pybind11::bytes(""); - } - if (cache_size) { - std::string calibration_cache_str(calibration_cache, - calibration_cache + cache_size); - std::ofstream cache_file; - cache_file.open(calibration_file, std::ios::out); - cache_file << calibration_cache_str; - delete calibration_cache; - calibration_cache = nullptr; - P2OLogger(verbose) << "TensorRT calibration cache path: " - << calibration_file << std::endl; - } - std::string onnx_proto(out, out + size); - delete out; - out = nullptr; - return pybind11::bytes(onnx_proto); - }); - m.def( - "optimize", - [](const std::string& model_path, const std::string& optimized_model_path, - const std::map>& shape_infos) { - ONNX_NAMESPACE::optimization::OptimizePaddle2ONNX( - model_path, optimized_model_path, shape_infos); - }); - m.def("convert_to_fp16", [](const std::string& fp32_model_path, - const std::string& fp16_model_path) { - paddle2onnx::optimization::Paddle2ONNXFP32ToFP16(fp32_model_path, - fp16_model_path); - }); -} -} // namespace paddle2onnx diff --git a/paddle2onnx/legacy/__init__.py b/paddle2onnx/legacy/__init__.py deleted file mode 100755 index 12737e8595d..00000000000 --- a/paddle2onnx/legacy/__init__.py +++ /dev/null @@ -1,128 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -from __future__ import absolute_import - -__version__ = "0.9.6" - -import paddle -from .convert import dygraph2onnx, program2onnx -from .op_mapper import register_op_mapper -from typing import TypeVar -from paddle2onnx.utils import logging -from paddle2onnx.legacy.op_mapper import OpMapper -from . import command - -OP_WITHOUT_KERNEL_SET = { - 'feed', 'fetch', 'recurrent', 'go', 'rnn_memory_helper_grad', - 'conditional_block', 'while', 'send', 'recv', 'listen_and_serv', - 'fl_listen_and_serv', 'ncclInit', 'select', 'checkpoint_notify', - 'gen_bkcl_id', 'c_gen_bkcl_id', 'gen_nccl_id', 'c_gen_nccl_id', - 'c_comm_init', 'c_sync_calc_stream', 'c_sync_comm_stream', - 'queue_generator', 'dequeue', 'enqueue', 'heter_listen_and_serv', - 'c_wait_comm', 'c_wait_compute', 'c_gen_hccl_id', 'c_comm_init_hccl', - 'copy_cross_scope' -} - - -def process_old_ops_desc(model): - for i in range(len(model.blocks[0].ops)): - if model.blocks[0].ops[i].type == "matmul": - if not model.blocks[0].ops[i].has_attr("head_number"): - model.blocks[0].ops[i]._set_attr("head_number", 1) - elif model.blocks[0].ops[i].type == "yolo_box": - if not model.blocks[0].ops[i].has_attr("iou_aware"): - model.blocks[0].ops[i]._set_attr("iou_aware", False) - if not model.blocks[0].ops[i].has_attr("iou_aware_factor"): - model.blocks[0].ops[i]._set_attr("iou_aware_factor", 0.5) - - -def get_all_registered_ops(save_file=None): - ops = list(OpMapper.OPSETS.keys()) - logging.warning("The number of all registered OPs is: {}".format(len(ops))) - if save_file is None: - return - with open(save_file, "w") as f: - logging.warning("All registered OPs will be written to the file: {}". - format(save_file)) - f.write("Total OPs num: {} \n".format(len(ops))) - for index in range(len(ops)): - op = ops[index] - f.write(str(index + 1) + ". " + op + "\n") - return - - -def run_convert(model, input_shape_dict=None, scope=None, opset_version=9): - paddle_version = paddle.__version__ - if isinstance(model, paddle.static.Program): - process_old_ops_desc(model) - if input_shape_dict is not None: - model_version = model.desc._version() - major_ver = model_version // 1000000 - minor_ver = (model_version - major_ver * 1000000) // 1000 - patch_ver = model_version - major_ver * 1000000 - minor_ver * 1000 - model_version = "{}.{}.{}".format(major_ver, minor_ver, patch_ver) - if model_version != paddle_version: - logging.warning( - "The model is saved by paddlepaddle v{}, but now your paddlepaddle is version of {}, this difference may cause error, it is recommend you reinstall a same version of paddlepaddle for this model". - format(model_version, paddle_version)) - for k, v in input_shape_dict.items(): - model.blocks[0].var(k).desc.set_shape(v) - for i in range(len(model.blocks[0].ops)): - if model.blocks[0].ops[i].type in OP_WITHOUT_KERNEL_SET: - continue - model.blocks[0].ops[i].desc.infer_shape(model.blocks[0].desc) - if scope is None: - scope = paddle.static.global_scope() - input_names = list() - output_vars = list() - for i in range(len(model.blocks[0].ops)): - if model.blocks[0].ops[i].type == "feed": - input_names.append(model.blocks[0].ops[i].output("Out")[0]) - if model.blocks[0].ops[i].type == "fetch": - output_vars.append(model.blocks[0].var(model.blocks[0].ops[i] - .input("X")[0])) - return program2onnx( - model, - scope, - save_file=None, - feed_var_names=input_names, - target_vars=output_vars, - opset_version=opset_version, - enable_onnx_checker=True) - elif isinstance(model, paddle.jit.TranslatedLayer): - process_old_ops_desc(model.program()) - model_version = model.program().desc._version() - major_ver = model_version // 1000000 - minor_ver = (model_version - major_ver * 1000000) // 1000 - patch_ver = model_version - major_ver * 1000000 - minor_ver * 1000 - model_version = "{}.{}.{}".format(major_ver, minor_ver, patch_ver) - if model_version != paddle_version: - logging.warning( - "The model is saved by paddlepaddle v{}, but now your paddlepaddle is version of {}, this difference may cause error, it is recommend you reinstall a same version of paddlepaddle for this model". - format(model_version, paddle_version)) - - if input_shape_dict is not None: - for k, v in input_shape_dict.items(): - model.program().blocks[0].var(k).desc.set_shape(v) - for i in range(len(model.program().blocks[0].ops)): - if model.program().blocks[0].ops[ - i].type in OP_WITHOUT_KERNEL_SET: - continue - model.program().blocks[0].ops[i].desc.infer_shape(model.program( - ).blocks[0].desc) - return dygraph2onnx(model, save_file=None, opset_version=opset_version) - else: - raise Exception( - "Only support model loaded from paddle.static.load_inference_model() or paddle.jit.load()" - ) diff --git a/paddle2onnx/legacy/command.py b/paddle2onnx/legacy/command.py deleted file mode 100755 index eb67b6d9fee..00000000000 --- a/paddle2onnx/legacy/command.py +++ /dev/null @@ -1,287 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import -from six import text_type as _text_type -import argparse -import ast -import sys -import os -import paddle.fluid as fluid -from paddle2onnx.utils import logging - - -def str2list(v): - if len(v) == 0: - return None - v = v.replace(" ", "") - v = eval(v) - return v - - -def arg_parser(): - parser = argparse.ArgumentParser() - parser.add_argument( - "--model_dir", - "-m", - type=_text_type, - default=None, - help="PaddlePaddle model directory, if params stored in single file, you need define '--model_filename' and 'params_filename'." - ) - parser.add_argument( - "--model_filename", - "-mf", - type=_text_type, - default=None, - help="PaddlePaddle model's network file name, which under directory seted by --model_dir" - ) - parser.add_argument( - "--params_filename", - "-pf", - type=_text_type, - default=None, - help="PaddlePaddle model's param file name(param files combined in single file), which under directory seted by --model_dir." - ) - parser.add_argument( - "--save_file", - "-s", - type=_text_type, - default=None, - help="file path to save onnx model") - parser.add_argument( - "--opset_version", - "-ov", - type=int, - default=9, - help="set onnx opset version to export") - parser.add_argument( - "--input_shape_dict", - "-isd", - type=_text_type, - default="None", - help="define input shapes, e.g --input_shape_dict=\"{'image':[1, 3, 608, 608]}\" or" \ - "--input_shape_dict=\"{'image':[1, 3, 608, 608], 'im_shape': [1, 2], 'scale_factor': [1, 2]}\"") - parser.add_argument( - "--enable_dev_version", - type=ast.literal_eval, - default=False, - help="whether to use new version of Paddle2ONNX which is under developing, default False" - ) - parser.add_argument( - "--enable_onnx_checker", - type=ast.literal_eval, - default=True, - help="whether check onnx model validity, default True") - parser.add_argument( - "--enable_paddle_fallback", - type=ast.literal_eval, - default=False, - help="whether use PaddleFallback for custom op, default is False") - parser.add_argument( - "--version", - "-v", - action="store_true", - default=False, - help="get version of paddle2onnx") - parser.add_argument( - "--output_names", - "-on", - type=str2list, - default=None, - help="define output names, e.g --output_names=\"[\"output1\"]\" or \ - --output_names=\"[\"output1\", \"output2\", \"output3\"]\" or \ - --output_names=\"{\"Paddleoutput\":\"Onnxoutput\"}\"") - parser.add_argument( - "--enable_auto_update_opset", - type=ast.literal_eval, - default=True, - help="whether enable auto_update_opset, default is True") - return parser - - -def c_paddle_to_onnx(model_file, - params_file="", - save_file=None, - opset_version=7, - auto_upgrade_opset=True, - verbose=True, - enable_onnx_checker=True, - enable_experimental_op=True, - enable_optimize=True): - import paddle2onnx.paddle2onnx_cpp2py_export as c_p2o - onnx_model_str = c_p2o.export( - model_file, params_file, opset_version, auto_upgrade_opset, verbose, - enable_onnx_checker, enable_experimental_op, enable_optimize) - if save_file is not None: - with open(save_file, "wb") as f: - f.write(onnx_model_str) - else: - return onnx_model_str - - -def program2onnx(model_dir, - save_file, - model_filename=None, - params_filename=None, - opset_version=9, - enable_onnx_checker=False, - operator_export_type="ONNX", - input_shape_dict=None, - output_names=None, - auto_update_opset=True): - try: - import paddle - except: - logging.error( - "paddlepaddle not installed, use \"pip install paddlepaddle\"") - - v0, v1, v2 = paddle.__version__.split('.') - if v0 == '0' and v1 == '0' and v2 == '0': - logging.warning("You are use develop version of paddlepaddle") - elif int(v0) <= 1 and int(v1) < 8: - raise ImportError("paddlepaddle>=1.8.0 is required") - - import paddle2onnx as p2o - # convert model save with 'paddle.fluid.io.save_inference_model' - if hasattr(paddle, 'enable_static'): - paddle.enable_static() - exe = fluid.Executor(fluid.CPUPlace()) - if model_filename is None and params_filename is None: - [program, feed_var_names, fetch_vars] = fluid.io.load_inference_model( - model_dir, exe) - else: - [program, feed_var_names, fetch_vars] = fluid.io.load_inference_model( - model_dir, - exe, - model_filename=model_filename, - params_filename=params_filename) - - OP_WITHOUT_KERNEL_SET = { - 'feed', 'fetch', 'recurrent', 'go', 'rnn_memory_helper_grad', - 'conditional_block', 'while', 'send', 'recv', 'listen_and_serv', - 'fl_listen_and_serv', 'ncclInit', 'select', 'checkpoint_notify', - 'gen_bkcl_id', 'c_gen_bkcl_id', 'gen_nccl_id', 'c_gen_nccl_id', - 'c_comm_init', 'c_sync_calc_stream', 'c_sync_comm_stream', - 'queue_generator', 'dequeue', 'enqueue', 'heter_listen_and_serv', - 'c_wait_comm', 'c_wait_compute', 'c_gen_hccl_id', 'c_comm_init_hccl', - 'copy_cross_scope' - } - if input_shape_dict is not None: - import paddle2onnx - paddle2onnx.legacy.process_old_ops_desc(program) - paddle_version = paddle.__version__ - model_version = program.desc._version() - major_ver = model_version // 1000000 - minor_ver = (model_version - major_ver * 1000000) // 1000 - patch_ver = model_version - major_ver * 1000000 - minor_ver * 1000 - model_version = "{}.{}.{}".format(major_ver, minor_ver, patch_ver) - if model_version != paddle_version: - logging.warning( - "The model is saved by paddlepaddle v{}, but now your paddlepaddle is version of {}, this difference may cause error, it is recommend you reinstall a same version of paddlepaddle for this model". - format(model_version, paddle_version)) - - for k, v in input_shape_dict.items(): - program.blocks[0].var(k).desc.set_shape(v) - for i in range(len(program.blocks[0].ops)): - if program.blocks[0].ops[i].type in OP_WITHOUT_KERNEL_SET: - continue - program.blocks[0].ops[i].desc.infer_shape(program.blocks[0].desc) - p2o.program2onnx( - program, - fluid.global_scope(), - save_file, - feed_var_names=feed_var_names, - target_vars=fetch_vars, - opset_version=opset_version, - enable_onnx_checker=enable_onnx_checker, - operator_export_type=operator_export_type, - auto_update_opset=auto_update_opset, - output_names=output_names) - - -def main(): - if len(sys.argv) < 2: - logging.info("Use \"paddle2onnx -h\" to print the help information") - logging.info( - "For more information, please follow our github repo below:") - logging.info("Github: https://github.com/PaddlePaddle/paddle2onnx.git") - return - - parser = arg_parser() - args = parser.parse_args() - - if args.version: - import paddle2onnx - logging.info("paddle2onnx-{} with python>=2.7, paddlepaddle>=1.8.0". - format(paddle2onnx.__version__)) - return - - assert args.model_dir is not None, "--model_dir should be defined while translating paddle model to onnx" - assert args.save_file is not None, "--save_file should be defined while translating paddle model to onnx" - - input_shape_dict = eval(args.input_shape_dict) - - operator_export_type = "ONNX" - if args.enable_paddle_fallback: - operator_export_type = "PaddleFallback" - - if args.output_names is not None: - if not isinstance(args.output_names, (list, dict)): - raise TypeError( - "The output_names should be 'list' or 'dict', but received type is %s." - % type(args.output_names)) - - if args.enable_dev_version: - if args.enable_paddle_fallback: - logging.warn( - "--enable_paddle_fallback is deprecated while --enable_dev_version=True." - ) - if args.output_names is not None: - logging.warn( - "--output_names is deprecated while --enable_dev_version=True.") - if input_shape_dict is not None: - logging.warn( - "--input_shape_dict is deprecated while --enable_dev_version=True." - ) - model_file = os.path.join(args.model_dir, args.model_filename) - if args.params_filename is None: - params_file = "" - else: - params_file = os.path.join(args.model_dir, args.params_filename) - return c_paddle_to_onnx( - model_file=model_file, - params_file=params_file, - save_file=args.save_file, - opset_version=args.opset_version, - auto_upgrade_opset=args.enable_auto_update_opset, - verbose=True, - enable_onnx_checker=args.enable_onnx_checker, - enable_experimental_op=True, - enable_optimize=True) - - program2onnx( - args.model_dir, - args.save_file, - args.model_filename, - args.params_filename, - opset_version=args.opset_version, - enable_onnx_checker=args.enable_onnx_checker, - operator_export_type=operator_export_type, - input_shape_dict=input_shape_dict, - output_names=args.output_names, - auto_update_opset=args.enable_auto_update_opset) - - -if __name__ == "__main__": - main() diff --git a/paddle2onnx/legacy/constant/__init__.py b/paddle2onnx/legacy/constant/__init__.py deleted file mode 100644 index 4361cadef1a..00000000000 --- a/paddle2onnx/legacy/constant/__init__.py +++ /dev/null @@ -1,15 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -from .constant import PRODUCER -from .constant import NodeDomain diff --git a/paddle2onnx/legacy/constant/constant.py b/paddle2onnx/legacy/constant/constant.py deleted file mode 100644 index b8e4fa4fe66..00000000000 --- a/paddle2onnx/legacy/constant/constant.py +++ /dev/null @@ -1,24 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -PRODUCER = 'PaddlePaddle' - -ONNX_HELPER_VERSION = '1.7.0' - - -class NodeDomain(): - ONNX = 'onnx' - PADDLE = 'paddle' - CUSTOM = 'custom' - RAW = 'raw' diff --git a/paddle2onnx/legacy/constant/dtypes.py b/paddle2onnx/legacy/constant/dtypes.py deleted file mode 100644 index 74fc6fe0637..00000000000 --- a/paddle2onnx/legacy/constant/dtypes.py +++ /dev/null @@ -1,84 +0,0 @@ -# Copyright (c) 2019 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -import numpy as np -import paddle.fluid.core as core -from onnx import helper -from onnx import TensorProto - -ONNX = TensorProto - -DTYPE_PADDLE_ONNX_MAP = { - TensorProto.FLOAT16: core.VarDesc.VarType.FP16, - TensorProto.FLOAT: core.VarDesc.VarType.FP32, - TensorProto.DOUBLE: core.VarDesc.VarType.FP64, - TensorProto.INT16: core.VarDesc.VarType.INT16, - TensorProto.INT32: core.VarDesc.VarType.INT32, - TensorProto.INT64: core.VarDesc.VarType.INT64, - TensorProto.BOOL: core.VarDesc.VarType.BOOL, - TensorProto.UINT8: core.VarDesc.VarType.UINT8, - core.VarDesc.VarType.FP16: TensorProto.FLOAT16, - core.VarDesc.VarType.FP32: TensorProto.FLOAT, - core.VarDesc.VarType.FP64: TensorProto.DOUBLE, - core.VarDesc.VarType.INT16: TensorProto.INT16, - core.VarDesc.VarType.INT32: TensorProto.INT32, - core.VarDesc.VarType.INT64: TensorProto.INT64, - core.VarDesc.VarType.BOOL: TensorProto.BOOL, - core.VarDesc.VarType.UINT8: TensorProto.UINT8, -} - -DTYPE_PADDLE_NUMPY_MAP = { - np.float32: core.VarDesc.VarType.FP32, - np.float64: core.VarDesc.VarType.FP64, - np.int16: core.VarDesc.VarType.INT16, - np.int32: core.VarDesc.VarType.INT32, - np.int64: core.VarDesc.VarType.INT64, - np.bool_: core.VarDesc.VarType.BOOL, - core.VarDesc.VarType.FP32: np.float32, - core.VarDesc.VarType.FP64: np.float64, - core.VarDesc.VarType.INT16: np.int16, - core.VarDesc.VarType.INT32: np.int32, - core.VarDesc.VarType.INT64: np.int64, - core.VarDesc.VarType.BOOL: np.bool_ -} - -DTYPE_PADDLE_STR_MAP = { - core.VarDesc.VarType.FP32: 'float32', - core.VarDesc.VarType.FP64: 'float64', - core.VarDesc.VarType.INT16: 'int16', - core.VarDesc.VarType.INT32: 'int32', - core.VarDesc.VarType.INT64: 'int64', - core.VarDesc.VarType.BOOL: 'bool', - 'float32': core.VarDesc.VarType.FP32, - 'float64': core.VarDesc.VarType.FP64, - 'int16': core.VarDesc.VarType.INT16, - 'int32': core.VarDesc.VarType.INT32, - 'int64': core.VarDesc.VarType.INT64, - 'bool': core.VarDesc.VarType.BOOL -} - -DTYPE_ONNX_STR_MAP = { - TensorProto.FLOAT: 'float32', - TensorProto.DOUBLE: 'float64', - TensorProto.INT16: 'int16', - TensorProto.INT32: 'int32', - TensorProto.INT64: 'int64', - TensorProto.BOOL: 'bool', - 'float32': TensorProto.FLOAT, - 'float64': TensorProto.DOUBLE, - 'int16': TensorProto.INT16, - 'int32': TensorProto.INT32, - 'int64': TensorProto.INT64, - 'bool': TensorProto.BOOL, -} diff --git a/paddle2onnx/legacy/constant/op_mapping_status.py b/paddle2onnx/legacy/constant/op_mapping_status.py deleted file mode 100644 index 0f9e074cda2..00000000000 --- a/paddle2onnx/legacy/constant/op_mapping_status.py +++ /dev/null @@ -1,19 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -OP_MAPPING_WAITTING = 0 -OP_MAPPING_NO_REGISTER = 1 -OP_MAPPING_NO_VERSION = 2 -OP_MAPPING_SUCCESSED = 3 -OP_MAPPING_FAILED = 4 diff --git a/paddle2onnx/legacy/convert.py b/paddle2onnx/legacy/convert.py deleted file mode 100755 index 86e7d4f6c57..00000000000 --- a/paddle2onnx/legacy/convert.py +++ /dev/null @@ -1,206 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import os -import six -import paddle -import numpy as np -from paddle.fluid.framework import Variable -from paddle2onnx.utils import check_model, logging -from paddle2onnx.legacy.graph import PaddleGraph, ONNXGraph -from paddle2onnx.legacy.passes import PassManager - - -def export_onnx(paddle_graph, - save_file, - opset_version=9, - enable_onnx_checker=False, - operator_export_type="ONNX", - verbose=False, - auto_update_opset=True, - output_names=None): - onnx_graph = ONNXGraph.build(paddle_graph, opset_version, - operator_export_type, verbose, - auto_update_opset) - onnx_graph = PassManager.run_pass( - onnx_graph, ['dumplicate_names_pass', 'inplace_node_pass']) - onnx_proto = onnx_graph.export_proto(enable_onnx_checker, output_names) - - if save_file is None: - return onnx_proto - - path, _ = os.path.split(save_file) - if path != '' and not os.path.isdir(path): - os.makedirs(path) - with open(save_file, 'wb') as f: - f.write(onnx_proto.SerializeToString()) - logging.info("ONNX model saved in {}".format(save_file)) - - -def program2onnx(program, - scope, - save_file, - feed_var_names=None, - target_vars=None, - opset_version=9, - enable_onnx_checker=False, - operator_export_type="ONNX", - auto_update_opset=True, - **configs): - from paddle import fluid - if hasattr(paddle, 'enable_static'): - paddle.enable_static() - if isinstance(program, paddle.fluid.framework.Program): - if feed_var_names is not None: - if isinstance(feed_var_names, six.string_types): - feed_var_names = [feed_var_names] - else: - if not (bool(feed_var_names) and all( - isinstance(name, six.string_types) - for name in feed_var_names)): - raise TypeError("'feed_var_names' should be a list of str.") - - if target_vars is not None: - if isinstance(target_vars, Variable): - target_vars = [target_vars] - else: - if not (bool(target_vars) and - all(isinstance(var, Variable) for var in target_vars)): - raise TypeError( - "'target_vars' should be a list of variable.") - - paddle_graph = PaddleGraph.build_from_program(program, feed_var_names, - target_vars, scope) - output_names = None - if 'output_names' in configs: - output_names = configs['output_names'] - if output_names is not None and not isinstance(output_names, - (list, dict)): - raise TypeError( - "The output_names should be 'list' or dict, but received type is %s." - % type(output_names)) - return export_onnx( - paddle_graph, - save_file, - opset_version, - enable_onnx_checker, - operator_export_type, - auto_update_opset=auto_update_opset, - output_names=output_names) - else: - raise TypeError( - "the input 'program' should be 'Program', but received type is %s." - % type(program)) - - -def dygraph2onnx(layer, save_file, input_spec=None, opset_version=9, **configs): - from paddle.nn import Layer - from paddle.fluid import core - from paddle.fluid.framework import Variable - from paddle.fluid.dygraph.dygraph_to_static import program_translator - from paddle.fluid import dygraph - if not isinstance(layer, Layer): - raise TypeError( - "the input 'layer' should be 'Layer', 'TranslatedLayer', but received type is %s." - % type(layer)) - - inner_input_spec = None - if input_spec is not None: - if not isinstance(input_spec, list): - raise TypeError( - "The input input_spec should be 'list', but received type is %s." - % type(input_spec)) - inner_input_spec = [] - for var in input_spec: - if isinstance(var, paddle.static.InputSpec): - inner_input_spec.append(var) - elif isinstance(var, (core.VarBase, Variable)): - inner_input_spec.append( - paddle.static.InputSpec.from_tensor(var)) - else: - raise TypeError( - "The element in input_spec list should be 'Variable' or `paddle.static.InputSpec`, but received element's type is %s." - % type(var)) - - output_spec = None - if 'output_spec' in configs: - output_spec = configs['output_spec'] - if not isinstance(output_spec, list): - raise TypeError( - "The output_spec should be 'list', but received type is %s." % - type(output_spec)) - for var in output_spec: - if not isinstance(var, (core.VarBase, Variable)): - raise TypeError( - "The element in output_spec list should be 'Variable', but received element's type is %s." - % type(var)) - - verbose = False - if 'verbose' in configs: - if isinstance(configs['verbose'], bool): - verbose = configs['verbose'] - else: - raise TypeError( - "The verbose should be 'bool', but received type is %s." % - type(configs['verbose'])) - - enable_onnx_checker = False - if 'enable_onnx_checker' in configs: - if isinstance(configs['enable_onnx_checker'], bool): - enable_onnx_checker = configs['enable_onnx_checker'] - else: - raise TypeError( - "The 'enable_onnx_checker' should be 'bool', but received type is %s." - % type(configs['enable_onnx_checker'])) - - operator_export_type = "ONNX" - enable_paddle_fallback = False - if 'enable_paddle_fallback' in configs: - if isinstance(configs['enable_paddle_fallback'], bool): - enable_paddle_fallback = configs['enable_paddle_fallback'] - if enable_paddle_fallback: - operator_export_type = "PaddleFallback" - else: - raise TypeError( - "The 'enable_paddle_fallback' should be 'bool', but received type is %s." - % type(configs['enable_paddle_fallback'])) - - paddle_graph = PaddleGraph.build_from_dygraph(layer, inner_input_spec, - output_spec) - - if 'get_paddle_graph' in configs: - return paddle_graph - - auto_update_opset = True - if 'auto_update_opset' in configs: - if isinstance(configs['auto_update_opset'], bool): - auto_update_opset = configs['auto_update_opset'] - else: - raise TypeError( - "The auto_update_opset should be 'bool', but received type is %s." - % type(configs['auto_update_opset'])) - - output_names = None - if 'output_names' in configs: - output_names = configs['output_names'] - if not isinstance(output_names, (list, dict)): - raise TypeError( - "The output_names should be 'list' or dict, but received type is %s." - % type(output_names)) - - return export_onnx(paddle_graph, save_file, opset_version, - enable_onnx_checker, operator_export_type, verbose, - auto_update_opset, output_names) diff --git a/paddle2onnx/legacy/graph/__init__.py b/paddle2onnx/legacy/graph/__init__.py deleted file mode 100644 index fb87f22543f..00000000000 --- a/paddle2onnx/legacy/graph/__init__.py +++ /dev/null @@ -1,17 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from .graph import Graph, Node -from .paddle_graph import PaddleGraph, PaddleNode -from .onnx_graph import ONNXGraph, ONNXNode diff --git a/paddle2onnx/legacy/graph/dygraph_helper.py b/paddle2onnx/legacy/graph/dygraph_helper.py deleted file mode 100755 index 1d4b6accbf9..00000000000 --- a/paddle2onnx/legacy/graph/dygraph_helper.py +++ /dev/null @@ -1,271 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import os -import numpy as np -import inspect -import six -import paddle -from paddle.fluid.io import _get_valid_program -from paddle.fluid.dygraph.dygraph_to_static.program_translator import ProgramTranslator, StaticFunction -from paddle.fluid.layers.utils import flatten, pack_sequence_as -from collections import OrderedDict -from paddle.fluid import dygraph -from paddle.fluid.dygraph.jit import declarative -from paddle.fluid import core -from paddle.fluid import layers -from paddle.nn import Layer -from paddle.fluid.framework import Block, ParamBase, Program, Variable, Parameter, program_guard -from paddle.fluid.dygraph.layers import Layer - -from paddle2onnx.utils import logging -from paddle2onnx.legacy.graph.graph_helper import prepend_feed_ops, append_fetch_ops - - -def _get_input_var_names(inputs, input_spec): - name_none_error = "The %s's name is None. " \ - "When using jit.save, please set InputSepc's name in " \ - "to_static(input_spec=[]) and jit.save(input_spec=[]) " \ - "and make sure they are consistent." - name_no_exists_error = "The tensor `%s` does not exists. " \ - "Please make sure the name of InputSpec or example Tensor " \ - "in input_spec is the same as the name of InputSpec in " \ - "`to_static` decorated on the Layer.forward method." - result_list = [] - input_var_names = [ - var.name for var in flatten(inputs) if isinstance(var, Variable) - ] - if input_spec is None: - # no prune - return input_var_names - else: - # fileter out non-tensor type spec infos. - input_spec = [ - spec for spec in input_spec - if isinstance(spec, paddle.static.InputSpec) - ] - - if len(input_spec) == len(input_var_names): - # no prune - result_list = input_var_names - # if input spec name not in input_var_names, only raise warning - for spec in input_spec: - if spec.name is None: - warnings.warn(name_none_error % spec) - elif spec.name not in input_var_names: - warnings.warn(name_no_exists_error % spec.name) - else: - # do nothing - pass - else: - # prune - for spec in input_spec: - if spec.name is None: - # name is None, the input_spec only can be InputSpec - raise ValueError(name_none_error % spec) - elif spec.name not in input_var_names: - # the input_spec can be `InputSpec` or `VarBase` - raise ValueError(name_no_exists_error % spec.name) - else: - result_list.append(spec.name) - - return result_list - - -def _get_output_vars(outputs, output_spec): - name_no_exists_error = "The tensor `%s` does not exists. " \ - "Please make sure the name of example Tensor " \ - "in configs.output_spec is the output tensor of " \ - "Layer.forward method." - result_list = [] - output_vars_dict = OrderedDict() - for var in flatten(outputs): - if isinstance(var, Variable): - output_vars_dict[var.name] = var - if output_spec is None: - result_list = output_vars_dict.values() - elif output_spec is not None and len(output_spec) == len(output_vars_dict): - result_list = output_vars_dict.values() - for var in output_spec: - if var.name not in output_vars_dict: - warnings.warn(name_no_exists_error % var.name) - else: - for var in output_spec: - if var.name not in output_vars_dict: - raise ValueError(name_no_exists_error % var.name) - else: - result_list.append(output_vars_dict[var.name]) - return result_list - - -@dygraph.base.switch_to_static_graph -def get_program(layer, input_spec, output_spec, **configs): - paddle.jit.set_verbosity(0) - prog_translator = ProgramTranslator() - if not prog_translator.enable_to_static: - raise RuntimeError( - "The Paddle2onnx doesn't work when setting ProgramTranslator.enable to False." - ) - - if not isinstance(layer, Layer): - raise TypeError( - "The input of paddle2onnx should be 'Layer', but received input type is %s." - % type(layer)) - - if isinstance(layer, paddle.DataParallel): - inner_layer = layer._layers - else: - inner_layer = layer - - # avoid change user given input_spec - inner_input_spec = None - if input_spec is not None: - for attr_func in dir(inner_layer): - static_func = getattr(inner_layer, attr_func, None) - if isinstance(static_func, - StaticFunction) and 'forward' != attr_func: - raise ValueError( - "If there are static functions other than 'forward' that need to be saved, the input 'input_spec' should be None, but received the type of 'input_spec' is %s." - % type(input_spec)) - - if not isinstance(input_spec, (list, tuple)): - raise TypeError( - "The input input_spec should be 'list', but received input_spec's type is %s." - % type(input_spec)) - inner_input_spec = [] - for var in flatten(input_spec): - if isinstance(var, paddle.static.InputSpec): - inner_input_spec.append(var) - elif isinstance(var, (core.VarBase, core.eager.Tensor, Variable)): - inner_input_spec.append( - paddle.static.InputSpec.from_tensor(var)) - else: - # NOTE(Aurelius84): Support non-Tensor type in `input_spec`. - inner_input_spec.append(var) - - extra_var_info = dict() - functions = dir(inner_layer) - for attr_func in functions: - static_func = getattr(inner_layer, attr_func, None) - if isinstance(static_func, StaticFunction): - concrete_program = static_func.concrete_program_specify_input_spec( - inner_input_spec) - elif 'forward' == attr_func: - # transform in jit.save, if input_spec is incomplete, declarative will throw error - # inner_input_spec is list[InputSpec], it should be packed with same structure - # as original input_spec here. - if inner_input_spec: - inner_input_spec = pack_sequence_as(input_spec, - inner_input_spec) - static_forward = declarative( - inner_layer.forward, input_spec=inner_input_spec) - concrete_program = static_forward.concrete_program - # the input_spec has been used in declarative, which is equal to - # @declarative with input_spec and jit.save without input_spec, - # avoid needless warning - inner_input_spec = None - else: - continue - - input_var_names = _get_input_var_names(concrete_program.inputs, - inner_input_spec) - - # NOTE(chenweihang): [ Get output variables ] - # the rule is like [ Get input variables name ]. For output var, - # we only support VarBase spec, and actually, we only need the - # var name of output, and we don't recommended to use output_spec - output_vars = _get_output_vars(concrete_program.outputs, output_spec) - - feeded_var_names = input_var_names - target_vars = output_vars - main_program = concrete_program.main_program.clone() - export_for_deployment = True - - if isinstance(feeded_var_names, six.string_types): - feeded_var_names = [feeded_var_names] - elif export_for_deployment: - if len(feeded_var_names) > 0: - # TODO(paddle-dev): polish these code blocks - if not (bool(feeded_var_names) and all( - isinstance(name, six.string_types) - for name in feeded_var_names)): - raise ValueError("'feed_var_names' should be a list of str.") - - if isinstance(target_vars, Variable): - target_vars = [target_vars] - elif export_for_deployment: - if not (bool(target_vars) and - all(isinstance(var, Variable) for var in target_vars)): - raise ValueError("'target_vars' should be a list of Variable.") - - main_program = _get_valid_program(main_program) - - # remind user to set auc_states to zeros if the program contains auc op - all_ops = main_program.global_block().ops - for op in all_ops: - # clear device of Op - device_attr_name = core.op_proto_and_checker_maker.kOpDeviceAttrName() - op._set_attr(device_attr_name, "") - if op.type == 'auc': - warnings.warn( - "please ensure that you have set the auc states to zeros before saving inference model" - ) - break - - with program_guard(main_program): - uniq_target_vars = [] - for i, var in enumerate(target_vars): - uniq_target_vars.append(var) - target_vars = uniq_target_vars - target_var_name_list = [var.name for var in target_vars] - - origin_program = main_program.clone() - - main_program = main_program.clone() - global_block = main_program.global_block() - need_to_remove_op_index = [] - for i, op in enumerate(global_block.ops): - op.desc.set_is_target(False) - if op.type == "feed" or op.type == "fetch": - need_to_remove_op_index.append(i) - - for index in need_to_remove_op_index[::-1]: - global_block._remove_op(index) - - main_program.desc.flush() - - main_program = main_program._prune_with_input( - feeded_var_names=feeded_var_names, targets=target_vars) - main_program = main_program._inference_optimize(prune_read_op=True) - fetch_var_names = [v.name for v in target_vars] - - for target_v in target_vars: - if not main_program.global_block().has_var(target_v.name): - main_program.global_block().create_var( - name=target_v.name, - shape=target_v.shape, - dtype=target_v.dtype, - persistable=target_v.persistable) - - prepend_feed_ops(main_program, feeded_var_names) - append_fetch_ops(main_program, fetch_var_names) - - main_program.desc._set_version() - paddle.fluid.core.save_op_version_info(main_program.desc) - - main_program._copy_dist_param_info_from(origin_program) - - return main_program, feeded_var_names, target_vars diff --git a/paddle2onnx/legacy/graph/graph.py b/paddle2onnx/legacy/graph/graph.py deleted file mode 100755 index a490e3d7502..00000000000 --- a/paddle2onnx/legacy/graph/graph.py +++ /dev/null @@ -1,287 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import os -import copy -import six -import collections -from paddle2onnx.legacy.constant import NodeDomain - - -class Node(object): - def __init__(self, - op_type, - inputs, - outputs, - attrs, - layer_name, - domain=NodeDomain.RAW): - self.domain = domain - self.type = op_type - self.attrs = attrs - self.layer_name = layer_name - self.set_inputs(inputs) - self.set_outputs(outputs) - - def __hash__(self): - return hash(self.layer_name) - - def __eq__(self, other): - if self.layer_name == other.layer_name: - return True - return False - - def __str__(self): - node_str = '' - attrs = '' - for key, value in self.attrs.items(): - attrs += ', ' + key + '=' + str(value) - node_str += " {} = {}::{}(inputs={}{}) \n".format( - self.outputs, self.domain, self.type, self.inputs, attrs) - return node_str - - def input(self, idx=None): - if idx is None: - return self.inputs - return self.inputs[idx] - - def output(self, idx=None): - if idx is None: - return self.outputs - return self.outputs[idx] - - def attr(self, name): - if name in self.attrs: - return self.attrs[name] - return None - - def set_inputs(self, inputs): - if isinstance(inputs, list): - self.inputs = [ - ipt.layer_name if isinstance(ipt, Node) else ipt - for ipt in inputs - ] - elif isinstance(inputs, six.string_types): - self.inputs = [inputs] - elif isinstance(inputs, Node): - self.inputs = [inputs.layer_name] - else: - raise TypeError( - 'Inputs of node must be type: list, Node, or String but got {}'. - format(type(inputs))) - - def set_outputs(self, outputs): - if isinstance(outputs, list): - self.outputs = [ - opt.layer_name if isinstance(opt, Node) else opt - for opt in outputs - ] - elif isinstance(outputs, six.string_types): - self.outputs = [outputs] - elif isinstance(outputs, Node): - self.outputs = [outputs.layer_name] - else: - raise TypeError( - 'Outputs of node must be type: list, Node, or String but got {}'. - format(type(outputs))) - - -class Graph(object): - def __init__(self): - self.parameters = {} - self.node_map = collections.OrderedDict() - self.input_nodes = list() - self.output_nodes = list() - self.op_type_count = dict() - - def __hash__(self): - return hash(self.id) - - def __eq__(self, other): - if self.id == other.id: - return True - return False - - def __str__(self): - graph_str = 'graph { \n' - for node in self.input_nodes: - graph_str += " input: {} \n".format(node.layer_name) - for node in self.output_nodes: - graph_str += " output: {} \n \n".format(node.layer_name) - for name, node in self.node_map.items(): - graph_str += node.__str__() - graph_str += ' }' - return graph_str - - def set_output_nodes(self, node_list): - if isinstance(node_list, list): - self.output_nodes = node_list - else: - raise TypeError( - 'output_nodes of Graph must be type: list, but got {}'.format( - type(node_list))) - - def set_node_map(self, node_map): - if isinstance(node_map, dict): - self.node_map = node_map - self.generate_topo_sort() - else: - raise TypeError('node_map of Graph must be type: list, but got {}'. - format(type(node_map))) - - def set_input_nodes(self, node_list): - if isinstance(node_list, list): - self.input_nodes = node_list - else: - raise TypeError( - 'input_nodes of Graph must be type: list, but got {}'.format( - type(node_list))) - - def set_parameters(self, parameters): - if isinstance(parameters, dict): - self.parameters = parameters - else: - raise TypeError( - 'parameters of Graph must be type: dict, but got {}'.format( - type(parameters))) - - def generate_node_name(self, op_type): - if op_type in self.op_type_count: - self.op_type_count[op_type] += 1 - else: - self.op_type_count[op_type] = 1 - # layer_name need follow https://github.com/onnx/onnx/blob/master/docs/OpConventions.md - layer_name = op_type + '_' + str(self.op_type_count[op_type] - 1) - return layer_name - - def insert_node(self, node): - if node.type not in ['feed', 'fetch']: - self.node_map[node.layer_name] = node - - def make_node(self, - op_type, - inputs=None, - outputs=None, - attrs=None, - layer_name=None, - domain=None, - **kw): - if layer_name is None: - layer_name = self.generate_node_name(op_type) - - if attrs is None: - attrs = kw - attrs.update(kw) - - if inputs is None: - inputs = [] - if outputs is None: - outputs = [layer_name] - node = Node(op_type, layer_name, inputs, outputs, attrs, domain) - self.insert_node(node) - return node - - def update_node(self, - node, - op_type=None, - inputs=None, - outputs=None, - attrs=None, - block=None, - move_to_end=True, - domain=None, - **kw): - if op_type is not None: - node.type = op_type - if inputs is not None: - node.set_inputs(inputs) - if outputs is not None: - node.set_outputs(outputs) - if attrs is None: - attrs = kw - attrs.update(kw) - node.attrs = attrs - if domain is not None: - node.domain = domain - if move_to_end: - self.node_map.pop(node.layer_name) - self.node_map[node.layer_name] = node - return node - - def get_node(self, name, copy=False): - if name not in self.node_map: - raise TypeError('Node with name:{} not in graph'.format(name)) - if copy: - node = copy.copy(self.node_map[name]) - else: - node = self.node_map[name] - return node - - def remove_node_by_name(self, name): - if name in self.node_map: - node = self.node_map.pop(name) - return node - raise TypeError('Node with name:{} not in graph'.format(name)) - - def remove_node(self, node): - if isinstance(node, Node): - node = self.remove_node_by_name(node.layer_name) - return node - else: - node = self.remove_node_by_name(node) - return node - - def get_output_nodes_of_node(self, node): - if node in self.edge_map: - return self.edge_map[node] - elif self.get_node(node.layer_name, copy=False): - return [] - else: - raise KeyError('Node with layer_name {} not in graph.egde_map'. - format(node.layer_name)) - - def get_adjacency_map(self): - adjacency_map = {} - for layer_name, current_node in self.node_map.items(): - inputs = current_node.inputs - for ipt in inputs: - for layer_name, node in self.node_map.items(): - if current_node == node: - continue - outputs = node.outputs - if ipt in outputs: - if node not in adjacency_map: - adjacency_map[node] = set([current_node]) - else: - adjacency_map[node].add(current_node) - return adjacency_map - - def get_topo_sort_list(self): - topo_sort_list = list() - adjacency_map = self.get_adjacency_map() - for layer_name, node in self.node_map.items(): - if node not in adjacency_map: - topo_sort_list.append(node) - idx = 0 - while idx < len(topo_sort_list): - current_node = topo_sort_list[idx] - for input_node, output_nodes in adjacency_map.items(): - if current_node in output_nodes: - adjacency_map[input_node].remove(current_node) - if len(adjacency_map[input_node]) == 0: - topo_sort_list.append(input_node) - idx += 1 - return topo_sort_list[::-1] diff --git a/paddle2onnx/legacy/graph/graph_helper.py b/paddle2onnx/legacy/graph/graph_helper.py deleted file mode 100644 index a62594b5cc7..00000000000 --- a/paddle2onnx/legacy/graph/graph_helper.py +++ /dev/null @@ -1,83 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import os -import paddle -import numpy as np -from paddle.fluid import core -from paddle.fluid.framework import Variable, program_guard -from paddle2onnx.utils import logging - - -def prepend_feed_ops(inference_program, - feed_target_names, - feed_holder_name='feed'): - if len(feed_target_names) == 0: - return - global_block = inference_program.global_block() - feed_var = global_block.create_var( - name=feed_holder_name, - type=core.VarDesc.VarType.FEED_MINIBATCH, - persistable=True) - for i, name in enumerate(feed_target_names): - if not global_block.has_var(name): - raise ValueError( - "The feed_var_names[{i}]: '{name}' doesn't exist in pruned inference program. " - "Please check whether '{name}' is a valid feed_var name, or remove it from feed_var_names " - "if '{name}' is not involved in the fetch_vars calculation.". - format( - i=i, name=name)) - out = global_block.var(name) - global_block._prepend_op( - type='feed', - inputs={'X': [feed_var]}, - outputs={'Out': [out]}, - attrs={'col': i}) - - -def append_fetch_ops(inference_program, - fetch_target_names, - fetch_holder_name='fetch'): - global_block = inference_program.global_block() - fetch_var = global_block.create_var( - name=fetch_holder_name, - type=core.VarDesc.VarType.FETCH_LIST, - persistable=True) - for i, name in enumerate(fetch_target_names): - global_block.append_op( - type='fetch', - inputs={'X': [name]}, - outputs={'Out': [fetch_var]}, - attrs={'col': i}) - - -def get_program(program, feed_var_names, fetch_vars): - global_block = program.global_block() - need_to_remove_op_index = [] - for i, op in enumerate(global_block.ops): - op.desc.set_is_target(False) - if op.type == "feed" or op.type == "fetch": - need_to_remove_op_index.append(i) - for index in need_to_remove_op_index[::-1]: - global_block._remove_op(index) - program.desc.flush() - program = program._prune_with_input( - feeded_var_names=feed_var_names, targets=fetch_vars) - program = program._inference_optimize(prune_read_op=True) - fetch_var_names = [v.name for v in fetch_vars] - prepend_feed_ops(program, feed_var_names) - append_fetch_ops(program, fetch_var_names) - return program diff --git a/paddle2onnx/legacy/graph/onnx_graph.py b/paddle2onnx/legacy/graph/onnx_graph.py deleted file mode 100755 index ac7ae9f5979..00000000000 --- a/paddle2onnx/legacy/graph/onnx_graph.py +++ /dev/null @@ -1,333 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import os -import copy -import collections -import numpy as np -from paddle2onnx.legacy.graph import Node, Graph -from paddle2onnx.legacy.constant import NodeDomain, PRODUCER, dtypes -from paddle2onnx.legacy.op_mapper import OpMapper -from onnx import helper -from paddle2onnx.utils import check_model, logging - - -class ONNXNode(Node): - def __init__(self, op_type, inputs, outputs, attrs, layer_name, domain): - super(ONNXNode, self).__init__(op_type, inputs, outputs, attrs, - layer_name, domain) - self.domain = domain - self.onnx_node = self.make_onnx_node() - - def make_onnx_constant_node(self): - dtype = self.attr('dtype') - value = self.attr('value') - if isinstance(value, list): - dims = (len(value), ) - elif value is None: - dims = () - value = [] - else: - dims = () - value = [value] - - if 'dims' in self.attrs: - dims = self.attrs['dims'] - - tensor = helper.make_tensor( - name=self.layer_name, data_type=dtype, dims=dims, vals=value) - - onnx_node = helper.make_node( - self.type, inputs=self.inputs, outputs=self.outputs, value=tensor) - - return onnx_node - - def make_onnx_node(self): - if self.type in ['Constant', 'ConstantOfShape']: - onnx_node = self.make_onnx_constant_node() - else: - onnx_node = helper.make_node( - self.type, - inputs=self.inputs, - outputs=self.outputs, - name=self.layer_name, - domain=self.domain, - **self.attrs) - return onnx_node - - -class ONNXGraph(Graph): - def __init__(self, - paddle_graph, - opset_version, - operator_export_type="ONNX", - block=None, - auto_update_opset=True): - super(ONNXGraph, self).__init__() - self.opset_version = opset_version - self.operator_export_type = operator_export_type - self.ctx = paddle_graph - self.custom = [] - if auto_update_opset: - self.update_opset_version() - - def __str__(self): - graph_str = 'graph { \n' - for node in self.input_nodes: - graph_str += " input: {} \n".format(node) - for node in self.output_nodes: - graph_str += " output: {} \n \n".format(node) - for name, node in self.node_map.items(): - graph_str += node.__str__() - graph_str += ' }' - return graph_str - - def make_node(self, - op_type, - inputs=[], - outputs=[], - attrs=None, - layer_name=None, - domain=None, - **kw): - if layer_name is None: - layer_name = self.generate_node_name(op_type) - - if domain is not None: - if domain not in self.custom: - self.custom.append(domain) - - if attrs is None: - attrs = kw - attrs.update(kw) - - if inputs is None: - inputs = [] - - real_outputs = None - if outputs is None: - real_outputs = [layer_name] - elif isinstance(outputs, int): - real_outputs = [] - for i in range(outputs): - real_outputs.append(self.generate_node_name(op_type)) - elif isinstance(outputs, list): - real_outputs = [] - if len(outputs) == 0: - real_outputs = [layer_name] - else: - for opt in outputs: - if isinstance(opt, Node): - real_outputs.append(opt.layer_name) - elif isinstance(opt, int): - real_outputs.append(self.generate_node_name(op_type)) - else: - real_outputs.append(opt) - else: - real_outputs = outputs - - node = ONNXNode(op_type, inputs, real_outputs, attrs, layer_name, - domain) - - self.insert_node(node) - if len(node.outputs) == 1: - return node.outputs[0] - else: - return node.outputs - - def update_node(self, - node, - op_type=None, - inputs=None, - outputs=None, - attrs=None, - **kw): - if op_type is None: - op_type = node.type - if inputs is None: - inputs = node.inputs - if outputs is None: - outputs = node.outputs - if attrs is None: - attrs = node.attrs - attrs.update(kw) - - node = ONNXNode(op_type, inputs, outputs, attrs, node.layer_name, - node.domain) - self.insert_node(node) - return node - - def build_parameters(self, parameters): - # build weight nodes - for name, param in parameters.items(): - weight = param['data'] - if weight is not np.ndarray: - weight = np.array(weight) - tensor = helper.make_tensor( - name=name, - dims=param['shape'], - data_type=dtypes.DTYPE_PADDLE_ONNX_MAP[param['dtype']], - vals=weight.flatten().tolist()) - node = helper.make_node( - 'Constant', inputs=[], outputs=[name], value=tensor) - self.parameters[name] = node - - def build_input_nodes(self, input_nodes): - # build input nodes - for ipt in input_nodes: - self.add_input_node(ipt.layer_name, - ipt.attr('shape'), ipt.attr('dtype')) - - def build_output_nodes(self, output_nodes): - # build output nodes - for opt in output_nodes: - self.add_output_node(opt.layer_name, - opt.attr('shape'), opt.attr('dtype')) - - def update_opset_version(self): - node_map = self.ctx.node_map - self.opset_version = OpMapper.get_recommend_opset_version( - node_map, self.opset_version) - - def build_op_nodes(self, node_map): - OpMapper.check_support_status(node_map, self.opset_version) - # build op nodes - for name, node in list(node_map.items()): - OpMapper.mapping(self, node, self.operator_export_type) - - def make_value_info(self, name, shape, dtype): - tensor_info = helper.make_tensor_value_info( - name=name, - shape=shape, - elem_type=dtypes.DTYPE_PADDLE_ONNX_MAP[dtype]) - return tensor_info - - def add_input_node(self, name, shape, dtype): - vi = self.make_value_info(name, shape, dtype) - self.input_nodes.append(vi) - - def add_output_node(self, name, shape, dtype): - vi = self.make_value_info(name, shape, dtype) - self.output_nodes.append(vi) - - def find_index(self, node_inout, name): - for i in range(len(node_inout)): - if node_inout[i] == name: - return i - return -1 - - def change_output_names(self, onnx_proto, output_names): - logging.info("The output of the ONNX model is set to: {}".format( - output_names)) - if isinstance(output_names, list): - assert len(output_names) == len( - onnx_proto.graph.output - ), "The provided output names are inconsistent with the output number of the onnx model when output_names is list" - origin_output_names = [] - for i in range(len(onnx_proto.graph.output)): - origin_output_names.append(onnx_proto.graph.output[i].name) - onnx_proto.graph.output[i].name = output_names[i] - - for i in range(len(onnx_proto.graph.node)): - node = onnx_proto.graph.node[i] - # Prevent changed names from being changed again - output_visited_node = [] - input_visited_node = [] - for j in range(len(origin_output_names)): - if origin_output_names[j] in node.output: - index = self.find_index(node.output, - origin_output_names[j]) - if index in output_visited_node: - continue - output_visited_node.append(index) - onnx_proto.graph.node[i].output[index] = output_names[j] - if origin_output_names[j] in node.input: - index = self.find_index(node.input, - origin_output_names[j]) - if index in input_visited_node: - continue - input_visited_node.append(index) - onnx_proto.graph.node[i].input[index] = output_names[j] - if isinstance(output_names, dict): - for i in range(len(onnx_proto.graph.output)): - for key, value in output_names.items(): - if onnx_proto.graph.output[i].name == key: - onnx_proto.graph.output[i].name = value - break - - for i in range(len(onnx_proto.graph.node)): - node = onnx_proto.graph.node[i] - # Prevent changed names from being changed again - output_visited_node = [] - input_visited_node = [] - for key, value in output_names.items(): - if key in node.output: - index = self.find_index(node.output, key) - if index in output_visited_node: - continue - output_visited_node.append(index) - onnx_proto.graph.node[i].output[index] = value - if key in node.input: - index = self.find_index(node.input, key) - if index in input_visited_node: - continue - input_visited_node.append(index) - onnx_proto.graph.node[i].input[index] = value - - return onnx_proto - - def export_proto(self, enable_onnx_checker=False, output_names=None): - - op_nodes = [node.onnx_node for node in self.node_map.values()] - weight_nodes = [node for node in self.parameters.values()] - - onnx_graph = helper.make_graph( - nodes=weight_nodes + op_nodes, - name='paddle-onnx', - initializer=[], - inputs=self.input_nodes, - outputs=self.output_nodes) - - opset_imports = [helper.make_opsetid("", self.opset_version)] - for custom_domain in self.custom: - opset_imports.append(helper.make_opsetid(custom_domain, 1)) - onnx_proto = helper.make_model( - onnx_graph, producer_name=PRODUCER, opset_imports=opset_imports) - if output_names is not None: - onnx_proto = self.change_output_names(onnx_proto, output_names) - - if enable_onnx_checker: - check_model(onnx_proto) - - return onnx_proto - - @staticmethod - def build(paddle_graph, - opset_version, - operator_export_type="ONNX", - verbose=False, - auto_update_opset=True): - onnx_graph = ONNXGraph( - paddle_graph, - opset_version=opset_version, - operator_export_type=operator_export_type, - auto_update_opset=auto_update_opset) - onnx_graph.build_parameters(paddle_graph.parameters) - onnx_graph.build_input_nodes(paddle_graph.input_nodes) - onnx_graph.build_output_nodes(paddle_graph.output_nodes) - onnx_graph.build_op_nodes(paddle_graph.node_map) - - return onnx_graph diff --git a/paddle2onnx/legacy/graph/paddle_graph.py b/paddle2onnx/legacy/graph/paddle_graph.py deleted file mode 100755 index 7e2f2be7c4c..00000000000 --- a/paddle2onnx/legacy/graph/paddle_graph.py +++ /dev/null @@ -1,303 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import os -import copy -import collections -import numpy as np -import paddle -from paddle import fluid -from paddle.fluid import dygraph -from paddle.fluid.framework import Operator -from paddle2onnx.legacy.graph import Node, Graph -from paddle2onnx.legacy.constant import NodeDomain -from paddle2onnx.utils import logging - - -class PaddleNode(Node): - def __init__(self, paddle_op, inputs, outputs, attrs, layer_name, block): - super(PaddleNode, self).__init__(paddle_op.type, inputs, outputs, attrs, - layer_name, NodeDomain.PADDLE) - self.paddle_op = paddle_op - self.block = block - - def __str__(self): - node_str = '' - attrs = '' - for key, value in self.attrs.items(): - if key == 'op_callstack': - continue - attrs += ', ' + key + '=' + str(value) - node_str += " {} = {}::{}(inputs={}{}) \n".format( - self.outputs, self.domain, self.type, self.inputs, attrs) - return node_str - - @property - def input_names(self): - return [name for name in self.inputs.keys()] - - @property - def output_names(self): - return [name for name in self.outputs.keys()] - - def input(self, name, idx=None): - if name not in self.inputs: - return None - if idx is None: - return self.inputs[name] - if len(self.inputs[name]) <= idx: - return None - return self.inputs[name][idx] - - def output(self, name, idx=None): - if idx is None: - return self.outputs[name] - return self.outputs[name][idx] - - def output_shape(self, name, idx): - return self.block.var(self.output(name, idx)).shape - - def input_shape(self, name, idx): - return self.block.var(self.input(name, idx)).shape - - def input_var(self, name, idx): - return self.block.var(self.input(name, idx)) - - def input_dtype(self, name, idx): - return self.block.var(self.input(name, idx)).dtype - - def output_dtype(self, name, idx): - return self.block.var(self.output(name, idx)).dtype - - def attr(self, name, default=None): - if name in self.attrs: - return self.attrs[name] - return default - - def set_inputs(self, inputs): - if isinstance(inputs, dict): - # input of node in paddle, which stored by dict - self.inputs = inputs - else: - raise TypeError('Inputs of node must be type: dict, but got {}'. - format(type(inputs))) - - def set_outputs(self, outputs): - if isinstance(outputs, dict): - # output of node in paddle, which stored by dict - self.outputs = outputs - else: - raise TypeError('Outputs of node must be type: dict, but got {}'. - format(type(outputs))) - - -class PaddleGraph(Graph): - def __init__(self, program, parameters, feed_var_names, fetch_vars): - super(PaddleGraph, self).__init__() - self.build_graph(program, parameters, feed_var_names, fetch_vars) - - def make_node(self, - op, - inputs=None, - outputs=None, - attrs=None, - block=None, - layer_name=None, - **kw): - if layer_name is None: - layer_name = self.generate_node_name(op.type) - - if attrs is None: - attrs = kw - attrs.update(kw) - - if inputs is None: - inputs = {} - if outputs is None: - outputs = {'Out': layer_name} - node = PaddleNode(op, inputs, outputs, attrs, layer_name, block) - self.insert_node(node) - return node - - def add_input_node(self, inputs, block=None): - for ipt in inputs: - # parse feed_names - layer_name = ipt - var = block.var(ipt) - attrs = {} - attrs['shape'] = var.shape - attrs['dtype'] = var.dtype - node = Node('feed', [], [layer_name], attrs, layer_name) - self.input_nodes.append(node) - - def add_output_node(self, outputs, block=None): - from paddle.fluid.framework import Variable - for opt in outputs: - # parse fetch_target_vars - layer_name = opt.name - attrs = {} - attrs['shape'] = opt.shape - attrs['dtype'] = opt.dtype - node = Node('fetch', [layer_name], [], attrs, layer_name) - self.output_nodes.append(node) - - def get_adjacency_map(self): - adjacency_map = {} - for layer_name, current_node in self.node_map.items(): - inputs = current_node.inputs.values() - inputs = [x for j in inputs for x in j] - for ipt in inputs: - for layer_name, node in self.node_map.items(): - if current_node == node: - continue - outputs = node.outputs.values() - outputs = [x for j in outputs for x in j] - if ipt in outputs: - if node not in adjacency_map: - adjacency_map[node] = set([current_node]) - else: - adjacency_map[node].add(current_node) - return adjacency_map - - def build_graph(self, - program, - parameters, - feed_var_names=None, - target_vars=None): - self.program = program - self.set_parameters(parameters) - self.add_input_node(feed_var_names, program.global_block()) - self.add_output_node(target_vars, program.global_block()) - for block in program.blocks: - for i, op in enumerate(block.ops): - if op.type in ['feed', 'fetch']: - continue - else: - inputs = {} - outputs = {} - for ipt in op.input_names: - inputs[ipt] = op.input(ipt) - for opt in op.output_names: - outputs[opt] = op.output(opt) - node = self.make_node(op, inputs, outputs, - op.all_attrs(), block) - - @staticmethod - def build_from_program(program, - feed_var_names=None, - fetch_vars=None, - scope=None): - parameters_dict = {} - vars = program.global_block().vars - for name in vars: - var = program.global_block().var(name) - if name.endswith('feed') or name.endswith('fetch'): - continue - if not var.persistable: - continue - parameters_dict[name] = { - 'data': np.array(scope.var(name).get_tensor()), - 'dtype': var.dtype, - 'shape': var.shape - } - - graph = PaddleGraph(program, parameters_dict, feed_var_names, - fetch_vars) - return graph - - @staticmethod - def build_from_dygraph(layer, input_spec=None, output_spec=None): - from paddle.nn import Layer - from paddle.fluid import core - from paddle.fluid.framework import Variable - from paddle2onnx.legacy.graph import dygraph_helper as dg_helper - if isinstance(layer, dygraph.TranslatedLayer): - program = layer.program() - parameters_dict = {} - pruned_vars = program.global_block().vars - for param in layer.parameters(): - if param.name.endswith('feed') or param.name.endswith('fetch'): - continue - if not param.persistable: - continue - if param.name in pruned_vars: - parameters_dict[param.name] = { - 'data': np.array(param.value().get_tensor()), - 'dtype': param.dtype, - 'shape': param.shape - } - for param in layer.buffers(): - if param.name.endswith('feed') or param.name.endswith('fetch'): - continue - if not param.value().get_tensor()._is_initialized(): - continue - if param.name in pruned_vars: - parameters_dict[param.name] = { - 'data': np.array(param.value().get_tensor()), - 'dtype': param.dtype, - 'shape': param.shape - } - if input_spec is not None: - logging.warning( - "Although input_spec is specified, TranslatedLayer is not support prune. An Complete network will be exported." - ) - input_spec = layer._input_spec() - if output_spec is not None: - logging.warning( - "Although output_spec is specified, TranslatedLayer is not support prune. An Complete network will be exported." - ) - feed_var_names = [ipt.name for ipt in layer._input_spec()] - fetch_vars = [ - program.global_block().var(opt.name) - for opt in layer._output_spec() - ] - graph = PaddleGraph(program, parameters_dict, feed_var_names, - fetch_vars) - return graph - elif isinstance(layer, Layer): - program, feed_var_names, fetch_vars = dg_helper.get_program( - layer, input_spec, output_spec) - parameters_dict = {} - pruned_vars = program.global_block().vars - for param in layer.parameters(): - if param.name.endswith('feed') or param.name.endswith('fetch'): - continue - if not param.persistable: - continue - if param.name in pruned_vars: - parameters_dict[param.name] = { - 'data': np.array(param.value().get_tensor()), - 'dtype': param.dtype, - 'shape': param.shape - } - for param in layer.buffers(): - if param.name.endswith('feed') or param.name.endswith('fetch'): - continue - if not param.value().get_tensor()._is_initialized(): - continue - if param.name in pruned_vars: - parameters_dict[param.name] = { - 'data': np.array(param.value().get_tensor()), - 'dtype': param.dtype, - 'shape': param.shape - } - graph = PaddleGraph(program, parameters_dict, feed_var_names, - fetch_vars) - return graph - else: - raise TypeError( - "The input Layer should be 'Layer' or 'TranslatedLayer', but received type is %s." - % type(layer)) diff --git a/paddle2onnx/legacy/op_mapper/__init__.py b/paddle2onnx/legacy/op_mapper/__init__.py deleted file mode 100644 index 78792e8f2d4..00000000000 --- a/paddle2onnx/legacy/op_mapper/__init__.py +++ /dev/null @@ -1,39 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -from .op_mapper import OpMapper, register_op_mapper, CustomPaddleOp, register_custom_paddle_op - -from . import nn -from . import math -from . import activation -from . import tensor -from . import logic -from . import search - -from .detection import yolo_box -from .detection import multiclass_nms -from .detection import prior_box -from .detection import density_prior_box -from .detection import box_coder -from .sequence import im2sequence - -from .custom_paddle_op import deformable_conv -from .custom_paddle_op import anchor_generator -from .custom_paddle_op import generate_proposals -from .custom_paddle_op import collect_fpn_proposals -from .custom_paddle_op import distribute_fpn_proposals -from .custom_paddle_op import box_clip -from .custom_paddle_op import grid_sampler diff --git a/paddle2onnx/legacy/op_mapper/activation.py b/paddle2onnx/legacy/op_mapper/activation.py deleted file mode 100755 index bda17016eed..00000000000 --- a/paddle2onnx/legacy/op_mapper/activation.py +++ /dev/null @@ -1,269 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import numpy as np -import math -from paddle2onnx.legacy.constant import dtypes -from paddle2onnx.legacy.op_mapper import OpMapper as op_mapper -from paddle2onnx.legacy.op_mapper import mapper_helper -import paddle - - -@op_mapper( - ['relu', 'tanh', 'log', 'sigmoid', 'sqrt'], - mapper_dict={ - 'relu': 'Relu', - 'tanh': 'Tanh', - 'log': 'Log', - 'sigmoid': 'Sigmoid', - 'sqrt': 'Sqrt', - }) -class ActivationOps(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - onnx_type = kw['mapper_dict'][node.type] - onnx_node = graph.make_node( - onnx_type, inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('silu') -class Silu(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - x = node.input('X')[0] - out = graph.make_node('Sigmoid', inputs=[x]) - graph.make_node('Mul', inputs=[x, out], outputs=node.output('Out')) - -@op_mapper('leaky_relu') -class LeakyRelu(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - onnx_node = graph.make_node( - 'LeakyRelu', - inputs=[node.input('X')[0]], - outputs=node.output('Out'), - alpha=node.attr('alpha')) - - -@op_mapper('softplus') -class Softplus(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - beta = node.attr('beta') - threshold = node.attr('threshold') - if np.isclose(beta, 1.0, 1e-06, 1e-06) and \ - np.isclose(threshold, 20.0, 1e-06, 1e-06): - onnx_node = graph.make_node( - 'Softplus', - inputs=[node.input('X')[0]], - outputs=node.output('Out')) - else: - raise Exception("[ERROR] Operator softplus " \ - "only supported while beta==1.0 and threshold==20.0") - - -@op_mapper('prelu') -class PRelu(): - support_opset_version_range = (9, 15) - - @classmethod - def opset_9(cls, graph, node, **kw): - slope_shape = node.input_shape('Alpha', 0) - input_shape = node.input_shape('X', 0) - - slope_node = node.input('Alpha')[0] - if len(input_shape) != len(slope_shape): - assert len( - slope_shape) == 1, "Slope shape is not expected for prelu" - broadcast_shape = [-1] + [1] * (len(input_shape) - 2) - broadcast_shape = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=broadcast_shape) - slope_node = graph.make_node( - 'Reshape', inputs=[node.input('Alpha')[0], broadcast_shape]) - x = node.input('X')[0] - x_dtype = node.input_dtype('X', 0) - slope_dtype = node.input_dtype('Alpha', 0) - if slope_dtype != paddle.float32: - slope_node = graph.make_node( - 'Cast', inputs=[slope_node], to=dtypes.ONNX.FLOAT) - if x_dtype != paddle.float32: - x = graph.make_node('Cast', inputs=[x], to=dtypes.ONNX.FLOAT) - onnx_node = graph.make_node('PRelu', inputs=[x, slope_node]) - graph.make_node( - 'Cast', - inputs=[onnx_node], - outputs=node.output('Out'), - to=dtypes.DTYPE_PADDLE_ONNX_MAP[x_dtype]) - else: - onnx_node = graph.make_node( - 'PRelu', inputs=[x, slope_node], outputs=node.output('Out')) - - -@op_mapper('relu6') -class Relu6(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - mapper_helper.clip_helper(graph, node, - node.input('X', 0), - node.attr('threshold'), 0.0, - node.output('Out', 0)) - - -@op_mapper('gelu') -class Gelu(): - support_opset_version_range = (9, 15) - - @classmethod - def opset_9(cls, graph, node, **kw): - input = node.input('X', 0) - x_dtype = node.input_dtype('X', 0) - # onnxruntime only support float32 Erf - if x_dtype != paddle.float32: - input = graph.make_node( - 'Cast', inputs=[input], to=dtypes.ONNX.FLOAT) - sqrt2 = graph.make_node( - 'Constant', dtype=dtypes.ONNX.FLOAT, value=[1.4142135623730951]) - zero_point_five = graph.make_node( - 'Constant', dtype=dtypes.ONNX.FLOAT, value=[0.5]) - one = graph.make_node('Constant', dtype=dtypes.ONNX.FLOAT, value=[1]) - x = graph.make_node('Div', inputs=[input, sqrt2]) - x = graph.make_node('Erf', inputs=x) - x = graph.make_node('Add', inputs=[x, one]) - x = graph.make_node('Mul', inputs=[input, x]) - if x_dtype != paddle.float32: - mul_node = graph.make_node('Mul', inputs=[x, zero_point_five]) - graph.make_node( - 'Cast', - inputs=[mul_node], - to=dtypes.DTYPE_PADDLE_ONNX_MAP[x_dtype], - outputs=node.output('Out')) - else: - graph.make_node( - 'Mul', inputs=[x, zero_point_five], outputs=node.output('Out')) - - -@op_mapper('selu') -class Selu(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_6(cls, graph, node, **kw): - graph.make_node( - 'Selu', - inputs=node.input('X'), - alpha=node.attr('alpha'), - gamma=node.attr('scale'), - outputs=node.output('Out')) - - -@op_mapper('hard_sigmoid') -class HardSigmoid(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - slope = node.attr('slope') - offset = node.attr('offset') - graph.make_node( - 'HardSigmoid', - inputs=node.input('X'), - outputs=node.output('Out'), - alpha=slope, - beta=offset) - - -@op_mapper('swish') -class Swish(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - x = node.input('X')[0] - if math.fabs(node.attr("beta") - 1.0) > 1e-05: - beta_node = graph.make_node( - 'Constant', - attrs={'dtype': dtypes.ONNX.FLOAT, - 'value': [node.attr('beta')]}) - x = graph.make_node( - 'Mul', inputs=[x, beta_node]) - sigmoid_node = graph.make_node('Sigmoid', inputs=[x]) - graph.make_node( - 'Mul', - inputs=[x, sigmoid_node], - outputs=node.output('Out')) - - -@op_mapper('mish') -class Mish(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - inputs = node.input('X', 0) - dtype = node.input_dtype("X", 0) - if dtype != paddle.float32: - inputs = graph.make_node( - 'Cast', inputs=[inputs], to=dtypes.ONNX.FLOAT) - dtype = paddle.float32 - threshold = node.attr('threshold') - assert np.fabs( - threshold - 20 - ) < 1e-4, "In mish OP, the threshold only supports 20, no other values are supported" - softplus_node = graph.make_node('Softplus', inputs=[inputs]) - tanh_node = graph.make_node('Tanh', inputs=[softplus_node]) - if node.input_dtype("X", 0) != paddle.float32: - mul_node = graph.make_node('Mul', inputs=[inputs, tanh_node]) - inputs = graph.make_node( - 'Cast', - inputs=[mul_node], - to=dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype("X", 0)], - outputs=node.output('Out')) - else: - graph.make_node( - 'Mul', inputs=[inputs, tanh_node], outputs=node.output('Out')) - - -@op_mapper('hard_swish') -class HardSwish(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - scale_node = graph.make_node( - 'Constant', - attrs={'dtype': dtypes.ONNX.FLOAT, - 'value': node.attr('scale')}) - offset_node = graph.make_node( - 'Constant', - attrs={'dtype': dtypes.ONNX.FLOAT, - 'value': node.attr('offset')}) - - node0 = graph.make_node('Add', inputs=[node.input('X')[0], offset_node]) - node1 = mapper_helper.clip_helper(graph, node, node0, - node.attr('threshold'), 0.0) - node2 = graph.make_node('Mul', inputs=[node.input('X')[0], node1]) - node3 = graph.make_node( - 'Div', inputs=[node2, scale_node], outputs=node.output('Out')) diff --git a/paddle2onnx/legacy/op_mapper/custom_paddle_op/__init__.py b/paddle2onnx/legacy/op_mapper/custom_paddle_op/__init__.py deleted file mode 100644 index 847ddc47ac8..00000000000 --- a/paddle2onnx/legacy/op_mapper/custom_paddle_op/__init__.py +++ /dev/null @@ -1,13 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. diff --git a/paddle2onnx/legacy/op_mapper/custom_paddle_op/anchor_generator.py b/paddle2onnx/legacy/op_mapper/custom_paddle_op/anchor_generator.py deleted file mode 100755 index 3f7517a7985..00000000000 --- a/paddle2onnx/legacy/op_mapper/custom_paddle_op/anchor_generator.py +++ /dev/null @@ -1,97 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import numpy as np -import paddle -from paddle.fluid import layers -from paddle2onnx.legacy.op_mapper import CustomPaddleOp, register_custom_paddle_op -from paddle2onnx.legacy.op_mapper import OpMapper as op_mapper -from paddle2onnx.legacy.op_mapper import mapper_helper - -class AnchorGenerator(CustomPaddleOp): - def __init__(self, node, **kw): - super(AnchorGenerator, self).__init__(node) - #self.x_shape = node.input_shape('Input', 0) - self.anchor_sizes = node.attr('anchor_sizes') - self.aspect_ratios = node.attr('aspect_ratios') - self.offset = node.attr('offset') - self.strides = node.attr('stride') - self.variances = node.attr('variances') - self.shapes = self.compute_shapes() - - def compute_shapes(self): - shapes = list() - for r in range(len(self.aspect_ratios)): - ar = self.aspect_ratios[r] - for s in range(len(self.anchor_sizes)): - anchor_size = self.anchor_sizes[s] - area = self.strides[0] * self.strides[1] - area_ratios = area / ar - base_w = np.floor(np.sqrt(area_ratios) + 0.5) - base_h = np.floor(base_w * ar + 0.5) - scale_w = anchor_size / self.strides[0] - scale_h = anchor_size / self.strides[1] - w = scale_w * base_w - h = scale_h * base_h - shapes.append([ - -0.5 * (w - 1), -0.5 * (h - 1), 0.5 * (w - 1), 0.5 * (h - 1) - ]) - return shapes - - def forward(self): - input_feature = self.input('Input', 0) - input_shape = paddle.shape(input_feature) - n, c, h, w = paddle.tensor.split(input_shape, num_or_sections=4) - x_ctr = paddle.arange(start=0, end=w, step=1, dtype=input_feature.dtype) - y_ctr = paddle.arange(start=0, end=h, step=1, dtype=input_feature.dtype) - x_ctr = x_ctr * self.strides[0] + self.offset * (self.strides[0] - 1) - y_ctr = y_ctr * self.strides[1] + self.offset * (self.strides[1] - 1) - tensor_one = paddle.ones(shape=[1], dtype='int64') - tensor_len_shape = paddle.full( - shape=[1], fill_value=len(self.shapes), dtype='int64') - x_ctr = paddle.reshape(x_ctr, shape=(1, -1)) - y_ctr = paddle.reshape(y_ctr, shape=(1, -1)) - x_ctr = paddle.tile(x_ctr, repeat_times=(h, tensor_one)) - y_ctr = paddle.tile(y_ctr, repeat_times=(w, tensor_one)) - y_ctr = paddle.transpose(y_ctr, perm=[1, 0]) - centers = paddle.stack([x_ctr, y_ctr], axis=-1) - centers = paddle.tensor.unsqueeze(centers, axis=[2]) - centers = paddle.tile(centers, repeat_times=(1, 1, len(self.shapes), 2)) - shape_tensor = paddle.assign(np.array(self.shapes).astype('float32')) - anchors = centers + shape_tensor - variance_tensor = paddle.assign( - np.asarray(self.variances).astype('float32')) - vars = paddle.reshape(variance_tensor, shape=[1, 1, 1, -1]) - vars = paddle.tile( - vars, repeat_times=(h, w, tensor_len_shape, tensor_one)) - return {'Anchors': [anchors], 'Variances': [vars]} - -@op_mapper('anchor_generator') -class Anchors_generator: - @classmethod - def opset_1(cls, graph, node, **kw): - node = graph.make_node( - 'anchor_generator', - inputs=node.input('Input'), - outputs=node.output('Anchors') + node.output('Variances'), - anchor_sizes = node.attr('anchor_sizes'), - aspect_ratios = node.attr('aspect_ratios'), - offset = node.attr('offset'), - strides = node.attr('stride'), - variances = node.attr('variances'), - domain = 'custom') - -register_custom_paddle_op('anchor_generator', AnchorGenerator) diff --git a/paddle2onnx/legacy/op_mapper/custom_paddle_op/box_clip.py b/paddle2onnx/legacy/op_mapper/custom_paddle_op/box_clip.py deleted file mode 100755 index e201c634835..00000000000 --- a/paddle2onnx/legacy/op_mapper/custom_paddle_op/box_clip.py +++ /dev/null @@ -1,56 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import numpy as np -import paddle -from paddle.fluid import layers -from paddle2onnx.legacy.op_mapper import CustomPaddleOp, register_custom_paddle_op -from paddle2onnx.legacy.op_mapper import OpMapper as op_mapper -from paddle2onnx.legacy.op_mapper import mapper_helper - -class BoxClip(CustomPaddleOp): - def __init__(self, node, **kw): - super(BoxClip, self).__init__(node) - - def forward(self): - input = self.input('Input', 0) - im_info = self.input('ImInfo', 0) - im_info = paddle.reshape(im_info, shape=[3]) - h, w, s = paddle.tensor.split(im_info, axis=0, num_or_sections=3) - tensor_one = paddle.full(shape=[1], dtype='float32', fill_value=1.0) - tensor_zero = paddle.full(shape=[1], dtype='float32', fill_value=0.0) - h = paddle.subtract(h, tensor_one) - w = paddle.subtract(w, tensor_one) - xmin, ymin, xmax, ymax = paddle.tensor.split( - input, axis=-1, num_or_sections=4) - xmin = paddle.maximum(paddle.minimum(xmin, w), tensor_zero) - ymin = paddle.maximum(paddle.minimum(ymin, h), tensor_zero) - xmax = paddle.maximum(paddle.minimum(xmax, w), tensor_zero) - ymax = paddle.maximum(paddle.minimum(ymax, h), tensor_zero) - cliped_box = paddle.concat([xmin, ymin, xmax, ymax], axis=-1) - - return {'Output': [cliped_box]} - -@op_mapper('box_clip') -class Boxclip: - @classmethod - def opset_1(cls, graph, node, **kw): - node = graph.make_node( - 'box_clip', - inputs=node.input('Input')+node.input('ImInfo'), - outputs=node.output('Output'), - domain = 'custom') -register_custom_paddle_op('box_clip', BoxClip) diff --git a/paddle2onnx/legacy/op_mapper/custom_paddle_op/collect_fpn_proposals.py b/paddle2onnx/legacy/op_mapper/custom_paddle_op/collect_fpn_proposals.py deleted file mode 100755 index 19119398006..00000000000 --- a/paddle2onnx/legacy/op_mapper/custom_paddle_op/collect_fpn_proposals.py +++ /dev/null @@ -1,55 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import numpy as np -import paddle -from paddle.fluid import layers -from paddle2onnx.legacy.op_mapper import CustomPaddleOp, register_custom_paddle_op -from paddle2onnx.legacy.op_mapper import OpMapper as op_mapper -from paddle2onnx.legacy.op_mapper import mapper_helper - - -class CollectFpnProposals(CustomPaddleOp): - def __init__(self, node, **kw): - super(CollectFpnProposals, self).__init__(node) - self.post_nms_top_n = node.attr('post_nms_topN') - - def forward(self): - multi_level_rois = self.input('MultiLevelRois') - multi_level_scores = self.input('MultiLevelScores') - multi_level_rois = paddle.concat(multi_level_rois, axis=0) - multi_level_scores = paddle.concat(multi_level_scores, axis=0) - proposal_num = paddle.shape(multi_level_scores)[0] - post_nms_top_n_tensor = paddle.assign( - np.array([self.post_nms_top_n]).astype('int32')) - k_candidate = paddle.concat([proposal_num, post_nms_top_n_tensor]) - k = paddle.min(k_candidate) - scores, index = paddle.topk(multi_level_scores, k=k, axis=0) - rois = paddle.gather(multi_level_rois, index, axis=0) - return {"FpnRois": [rois]} - -@op_mapper('collect_fpn_proposals') -class Collectfpnproposals: - @classmethod - def opset_1(cls, graph, node, **kw): - node = graph.make_node( - 'collect_fpn_proposals', - inputs=node.input('MultiLevelRois')+ node.input('MultiLevelScores'), - outputs=node.output('FpnRois'), - post_nms_top_n = node.attr('post_nms_topN'), - domain = 'custom') - -register_custom_paddle_op('collect_fpn_proposals', CollectFpnProposals) diff --git a/paddle2onnx/legacy/op_mapper/custom_paddle_op/deformable_conv.py b/paddle2onnx/legacy/op_mapper/custom_paddle_op/deformable_conv.py deleted file mode 100755 index 0042cd758d3..00000000000 --- a/paddle2onnx/legacy/op_mapper/custom_paddle_op/deformable_conv.py +++ /dev/null @@ -1,296 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import numpy as np -import paddle -from paddle.fluid import layers -from paddle2onnx.legacy.op_mapper import CustomPaddleOp, register_custom_paddle_op -from paddle2onnx import utils -from paddle2onnx.legacy.constant import dtypes -from paddle2onnx.legacy.op_mapper import OpMapper as op_mapper -from paddle2onnx.legacy.op_mapper import mapper_helper - - -class DeformConv2d(CustomPaddleOp): - def check_attribute(self, node): - utils.compare_attr_between_dims( - node.attr('strides'), (0, 1), 'strides', 'equal') - utils.compare_attr_between_dims( - node.attr('paddings'), (0, 1), 'paddings', 'equal') - utils.compare_attr_between_dims( - node.input_shape('Offset', 0), (2, 3), 'Offset', 'equal') - utils.compare_attr( - node.attr('deformable_groups'), 1, 'deformable_groups', 'equal') - - def __init__(self, node, **kw): - super(DeformConv2d, self).__init__(node) - self.check_attribute(node) - self.in_channel = node.input_shape('Input', 0)[1] - self.offset_channel = node.input_shape('Offset', 0)[1] - self.stride = node.attr('strides')[0] - self.padding = node.attr('paddings') - if len(self.padding) == 2: - self.padding += self.padding - self.groups = node.attr('groups') - self.dilation = node.attr('dilations')[0] - self.padded_x_h = node.input_shape('Input', 0)[2] - self.padded_x_w = node.input_shape('Input', 0)[3] - if self.padded_x_h > 0: - self.padded_x_h = self.padded_x_h + self.padding[0] + self.padding[1] - if self.padded_x_w > 0: - self.padded_x_w = self.padded_x_w + self.padding[2] + self.padding[3] - - self.kernel_size = node.input_shape('Filter', 0)[2] - self.N = self.kernel_size**2 - self.num_filters = node.input_shape('Filter', 0)[0] - - def forward(self): - input = self.input('Input', 0) - weight = self.input('Filter', 0) - mask = self.input('Mask', 0) - offset = self.input('Offset', 0) - - input = layers.pad2d(input, self.padding) - input_shape = paddle.shape(input) - if self.padded_x_h < 0 or self.padded_x_w < 0: - self.padded_x_h = input_shape[2] - self.padded_x_w = input_shape[3] - - offset_x = paddle.strided_slice( - offset, - axes=[1], - starts=[0], - ends=[self.offset_channel], - strides=[2]) - offset_y = paddle.strided_slice( - offset, - axes=[1], - starts=[1], - ends=[self.offset_channel], - strides=[2]) - offset = paddle.concat([offset_x, offset_y], axis=1) - offset_shape = paddle.shape(offset) - offset_h = offset_shape[2] - offset_w = offset_shape[3] - - coordinate = self.get_offset_coordinate(offset, 'float32', offset_shape) - - coordinate = coordinate.transpose((0, 2, 3, 1)) - coord_lt, coord_rb, coord_lb, coord_rt = self.get_bilinear_corner_coordinate( - coordinate, self.padded_x_h, self.padded_x_w) - - # clip coordinate - coordinate = paddle.concat( - [ - paddle.clip(coordinate[:, :, :, :self.N], 0, - self.padded_x_h - 1), - paddle.clip(coordinate[:, :, :, self.N:], 0, - self.padded_x_w - 1) - ], - axis=-1) - - cof_lt, cof_rb, cof_lb, cof_rt = self.get_bilinear_coefficient( - coord_lt, coord_rb, coord_lb, coord_rt, coordinate) - - feature_lt = self.get_feature_by_coordinate(input, coord_lt, offset_h, - offset_w, self.padded_x_w) - feature_rb = self.get_feature_by_coordinate(input, coord_rb, offset_h, - offset_w, self.padded_x_w) - feature_lb = self.get_feature_by_coordinate(input, coord_lb, offset_h, - offset_w, self.padded_x_w) - feature_rt = self.get_feature_by_coordinate(input, coord_rt, offset_h, - offset_w, self.padded_x_w) - - feature_after_deformation = paddle.unsqueeze(cof_lt, 1) * feature_lt + \ - paddle.unsqueeze(cof_rb, 1) * feature_rb + \ - paddle.unsqueeze(cof_lb, 1) * feature_lb + \ - paddle.unsqueeze(cof_rt, 1) * feature_rt - - # modulation - if mask is not None: - mask = paddle.transpose(mask, (0, 2, 3, 1)) - mask = paddle.unsqueeze(mask, 1) - mask = paddle.tile(mask, [1, self.in_channel, 1, 1, 1]) - feature_after_deformation *= mask - - feature_after_deformation = self.reshape_feature( - feature_after_deformation, offset_h, offset_w) - - out = paddle.nn.functional.conv2d( - feature_after_deformation, - weight, - stride=self.kernel_size, - groups=self.groups) - - return {'Output': [out]} - - def get_offset_coordinate(self, offset, dtype, offset_shape): - kernel_grid_origin_x = paddle.arange( - 0, - self.kernel_size + (self.kernel_size - 1) * (self.dilation - 1), - step=self.dilation, - dtype=dtype) - kernel_grid_origin_x = kernel_grid_origin_x.unsqueeze(1) - kernel_grid_origin_x = paddle.tile(kernel_grid_origin_x, - [1, self.kernel_size]) - kernel_grid_origin_y = paddle.arange( - 0, - self.kernel_size + (self.kernel_size - 1) * (self.dilation - 1), - step=self.dilation, - dtype=dtype) - kernel_grid_origin_y = kernel_grid_origin_y.unsqueeze(0) - kernel_grid_origin_y = paddle.tile(kernel_grid_origin_y, - [self.kernel_size, 1]) - kernel_grid_origin_x = paddle.reshape(kernel_grid_origin_x, [-1]) - kernel_grid_origin_y = paddle.reshape(kernel_grid_origin_y, [-1]) - kernel_grid_origin = paddle.concat( - [kernel_grid_origin_x, kernel_grid_origin_y], -1) - kernel_grid_origin = paddle.reshape(kernel_grid_origin, - (1, 2 * self.N, 1, 1)) - - kernel_offset_x = paddle.arange( - 0, offset_shape[2] * self.stride, step=self.stride, dtype=dtype) - kernel_offset_x = kernel_offset_x.unsqueeze(1) - kernel_offset_x = paddle.expand(kernel_offset_x, offset_shape[2:]) - kernel_offset_y = paddle.arange( - 0, offset_shape[3] * self.stride, step=self.stride, dtype=dtype) - kernel_offset_y = kernel_offset_y.unsqueeze(0) - kernel_offset_y = paddle.expand(kernel_offset_y, offset_shape[2:]) - kernel_offset_x = kernel_offset_x.unsqueeze([0, 1]) - kernel_offset_x = paddle.tile(kernel_offset_x, (1, self.N, 1, 1)) - kernel_offset_y = kernel_offset_y.unsqueeze([0, 1]) - kernel_offset_y = paddle.tile(kernel_offset_y, (1, self.N, 1, 1)) - - kernel_offset = paddle.concat([kernel_offset_x, kernel_offset_y], 1) - offset = offset + paddle.cast(kernel_offset, 'float32') + paddle.cast( - kernel_grid_origin, 'float32') - - return offset - - def get_bilinear_corner_coordinate(self, coord, padded_h, padded_w): - coord_lt = coord.floor() - coord_rb = coord_lt + 1 - coord_lt = paddle.cast( - paddle.concat( - [ - paddle.clip(coord_lt[:, :, :, :self.N], 0, padded_h - 1), - paddle.clip(coord_lt[:, :, :, self.N:], 0, padded_w - 1) - ], - axis=-1), - dtype='int64') - coord_rb = paddle.cast( - paddle.concat( - [ - paddle.clip(coord_rb[:, :, :, :self.N], 0, padded_h - 1), - paddle.clip(coord_rb[:, :, :, self.N:], 0, padded_w - 1) - ], - axis=-1), - dtype='int64') - coord_lb = paddle.concat( - [coord_lt[:, :, :, :self.N], coord_rb[:, :, :, self.N:]], axis=-1) - coord_rt = paddle.concat( - [coord_rb[:, :, :, :self.N], coord_lt[:, :, :, self.N:]], axis=-1) - - return coord_lt, coord_rb, coord_lb, coord_rt - - def get_bilinear_coefficient(self, coord_lt, coord_rb, coord_lb, coord_rt, - p): - cof_lt = (1 + (paddle.cast( - coord_lt[:, :, :, :self.N], dtype='float32') - p[:, :, :, :self.N]) - ) * (1 + paddle.cast( - coord_lt[:, :, :, self.N:], dtype='float32') - - p[:, :, :, self.N:]) - cof_rb = (1 - (paddle.cast( - coord_rb[:, :, :, :self.N], dtype='float32') - p[:, :, :, :self.N]) - ) * (1 - (paddle.cast( - coord_rb[:, :, :, self.N:], dtype='float32') - - p[:, :, :, self.N:])) - cof_lb = (1 + (paddle.cast( - coord_lb[:, :, :, :self.N], dtype='float32') - p[:, :, :, :self.N]) - ) * (1 - (paddle.cast( - coord_lb[:, :, :, self.N:], dtype='float32') - - p[:, :, :, self.N:])) - cof_rt = (1 - (paddle.cast( - coord_rt[:, :, :, :self.N], dtype='float32') - p[:, :, :, :self.N]) - ) * (1 + paddle.cast( - coord_rt[:, :, :, self.N:], dtype='float32') - - p[:, :, :, self.N:]) - - return cof_lt, cof_rb, cof_lb, cof_rt - - def get_feature_by_coordinate(self, x, coord, offset_h, offset_w, - padded_x_w): - x = paddle.reshape(x, [0, 0, -1]) - index = paddle.cast( - coord[:, :, :, :self.N] * padded_x_w, - dtype='int64') + coord[:, :, :, self.N:] # offset_x*w + offset_y - index = paddle.unsqueeze(index, 1) - index = paddle.tile(index, [1, self.in_channel, 1, 1, 1]) - index = paddle.reshape(index, (0, 0, -1)) - x_range = list(range(3)) - dim = 2 - x_range[0] = dim - x_range[dim] = 0 - x_swaped = paddle.transpose(x, perm=x_range) - index_range = list(range(3)) - index_range[0] = dim - index_range[dim] = 0 - index_swaped = paddle.transpose(index, perm=index_range) - x_shape = layers.shape(x_swaped) - index_shape = layers.shape(index_swaped) - prod = paddle.prod(x_shape[1:], keepdim=True) - x_swaped_flattend = paddle.reshape(x_swaped, [-1]) - index_swaped_flattend = paddle.reshape(index_swaped, [-1]) - index_swaped_flattend *= prod - bias = paddle.arange(start=0, end=prod, step=1, dtype='float32') - bias = paddle.tile(bias, index_shape[0]) - index_swaped_flattend += bias - gathered = paddle.gather(x_swaped_flattend, index_swaped_flattend) - gathered = paddle.reshape(gathered, layers.shape(index_swaped)) - x_offset = paddle.transpose(gathered, perm=x_range) - x_offset = paddle.reshape( - x_offset, (-1, self.in_channel, offset_h, offset_w, self.N)) - return x_offset - - def reshape_feature(self, x_offset, offset_h, offset_w): - x_offset = paddle.concat( - [ - paddle.reshape(x_offset[:, :, :, :, s:s + self.kernel_size], ( - -1, self.in_channel, offset_h, offset_w * self.kernel_size)) - for s in range(0, self.N, self.kernel_size) - ], - axis=-1) - x_offset = paddle.reshape(x_offset, (-1, self.in_channel, - offset_h * self.kernel_size, - offset_w * self.kernel_size)) - return x_offset - -@op_mapper('deformable_conv') -class Deformconv2d: - @classmethod - def opset_1(cls, graph, node, **kw): - node = graph.make_node( - 'deformable_conv', - inputs=node.input('Input')+node.input('Filter')+node.input('Mask')+node.input('Offset'), - outputs=node.output('Output'), - stride = node.attr('strides'), - padding = node.attr('paddings'), - groups = node.attr('groups'), - dilation = node.attr('dilations'), - deformable_groups = node.attr('deformable_groups'), - domain = 'custom') - -register_custom_paddle_op('deformable_conv', DeformConv2d) diff --git a/paddle2onnx/legacy/op_mapper/custom_paddle_op/distribute_fpn_proposals.py b/paddle2onnx/legacy/op_mapper/custom_paddle_op/distribute_fpn_proposals.py deleted file mode 100755 index eaefd0d4f05..00000000000 --- a/paddle2onnx/legacy/op_mapper/custom_paddle_op/distribute_fpn_proposals.py +++ /dev/null @@ -1,100 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import numpy as np -import paddle -from paddle.fluid import layers -from paddle2onnx.legacy.op_mapper import CustomPaddleOp, register_custom_paddle_op -from paddle2onnx.legacy.op_mapper import OpMapper as op_mapper -from paddle2onnx.legacy.op_mapper import mapper_helper - - -class DistributeFpnProposals(CustomPaddleOp): - def __init__(self, node, **kw): - super(DistributeFpnProposals, self).__init__(node) - self.max_level = node.attr('max_level') - self.min_level = node.attr('min_level') - self.refer_level = node.attr('refer_level') - self.refer_scale = node.attr('refer_scale') - self.pixel_offset = node.attr('pixel_offset') - - def bbox_area(self, boxes): - offset = 1 if self.pixel_offset else 0 - xmin, ymin, xmax, ymax = paddle.tensor.split( - boxes, axis=1, num_or_sections=4) - width = xmax - xmin + offset - height = ymax - ymin + offset - areas = width * height - return areas - - def forward(self): - fpn_rois = self.input('FpnRois', 0) - areas = self.bbox_area(fpn_rois) - scale = paddle.sqrt(areas) - num_level = self.max_level - self.min_level + 1 - target_level = paddle.log(scale / self.refer_scale + 1e-06) / np.log(2) - target_level = paddle.floor(self.refer_level + target_level) - target_level = paddle.clip( - target_level, min=self.min_level, max=self.max_level) - - rois = list() - rois_idx_order = list() - rois_num_per_level = list() - - for level in range(self.min_level, self.max_level + 1): - level_tensor = paddle.full_like(target_level, fill_value=level) - res = paddle.equal(target_level, level_tensor) - res = paddle.squeeze(res, axis=1) - res = paddle.cast(res, dtype='int32') - index = paddle.nonzero(res) - roi = paddle.gather(fpn_rois, index, axis=0) - rois.append(roi) - rois_idx_order.append(index) - rois_num_per_level.append(paddle.shape(roi)[0]) - rois_idx_order = paddle.concat(rois_idx_order, axis=0) - size = paddle.shape(rois_idx_order)[0] - _, rois_idx_restore = paddle.topk( - rois_idx_order, axis=0, sorted=True, largest=False, k=size) - - rois_idx_restore = paddle.cast(rois_idx_restore, dtype='int32') - if len(self.input('RoisNum')) > 0: - # trick: to keep rois num - rois_num_per_level[0] += self.input('RoisNum', 0) * 0 - return { - 'MultiFpnRois': rois, - 'RestoreIndex': [rois_idx_restore], - 'MultiLevelRoIsNum': rois_num_per_level - } - else: - return {'MultiFpnRois': rois, 'RestoreIndex': [rois_idx_restore]} - - -@op_mapper('distribute_fpn_proposals') -class Distributefpnproposals: - @classmethod - def opset_1(cls, graph, node, **kw): - node = graph.make_node( - 'distribute_fpn_proposals', - inputs=node.input('FpnRois'), - outputs=node.output('MultiFpnRois') + node.output('RestoreIndex'), - max_level=node.attr('max_level'), - min_level=node.attr('min_level'), - refer_level=node.attr('refer_level'), - refer_scale=node.attr('refer_scale'), - domain='custom') - - -register_custom_paddle_op('distribute_fpn_proposals', DistributeFpnProposals) diff --git a/paddle2onnx/legacy/op_mapper/custom_paddle_op/generate_proposals.py b/paddle2onnx/legacy/op_mapper/custom_paddle_op/generate_proposals.py deleted file mode 100755 index f1ae448d425..00000000000 --- a/paddle2onnx/legacy/op_mapper/custom_paddle_op/generate_proposals.py +++ /dev/null @@ -1,223 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import numpy as np -import paddle -import math -from paddle.fluid import layers -from paddle2onnx.legacy.op_mapper import CustomPaddleOp, register_custom_paddle_op -from paddle2onnx.legacy.op_mapper import OpMapper as op_mapper -from paddle2onnx.legacy.op_mapper import mapper_helper - -BBOX_CLIP_DEFAULT = math.log(1000.0 / 16.0) - - -class GenerateProposals(CustomPaddleOp): - def __init__(self, node, **kw): - paddle.enable_static() - super(GenerateProposals, self).__init__(node) - self.eta = node.attr('eta') - self.min_size = node.attr('min_size') - self.nms_thresh = node.attr('nms_thresh') - self.post_nms_topN = node.attr('post_nms_topN') - self.pre_nms_topN = node.attr('pre_nms_topN') - self.type = node.type - if self.type == 'generate_proposals_v2': - self.pixel_offset = node.attr('pixel_offset') - else: - self.pixel_offset = True - - def filter_boxes(self, boxes, im_w, im_h, im_s, min_size): - min_size = max(min_size, 1.0) - xmin, ymin, xmax, ymax = paddle.tensor.split( - boxes, axis=1, num_or_sections=4) - x_ctr = (xmax + xmin) / 2 + 0.5 - y_ctr = (ymax + ymin) / 2 + 0.5 - ws = (xmax - xmin) / im_s + 1 - hs = (ymax - ymin) / im_s + 1 - - min_size = np.asarray([min_size], dtype='float32') - min_size = paddle.assign(min_size) - valid_flag_ws = paddle.greater_equal(ws, min_size) - valid_flag_hs = paddle.greater_equal(hs, min_size) - valid_flag_x = paddle.less_equal(x_ctr, im_w) - valid_flag_y = paddle.less_equal(y_ctr, im_h) - valid_flag = paddle.logical_and(valid_flag_ws, valid_flag_hs) - valid_flag = paddle.logical_and(valid_flag, valid_flag_x) - valid_flag = paddle.logical_and(valid_flag, valid_flag_y) - valid_flag = paddle.squeeze(valid_flag, axis=1) - valid_inds = paddle.nonzero(valid_flag) - - return valid_inds - - def filter_boxes_v2(self, boxes, im_w, im_h, min_size, pixel_offset=True): - min_size = max(min_size, 1.0) - xmin, ymin, xmax, ymax = paddle.tensor.split( - boxes, axis=1, num_or_sections=4) - - offset = 1 if pixel_offset else 0 - ws = (xmax - xmin) + offset - hs = (ymax - ymin) + offset - - min_size = np.asarray([min_size], dtype='float32') - min_size = paddle.assign(min_size) - valid_flag_ws = paddle.greater_equal(ws, min_size) - valid_flag_hs = paddle.greater_equal(hs, min_size) - valid_flag = paddle.logical_and(valid_flag_ws, valid_flag_hs) - if pixel_offset: - x_ctr = xmin + ws / 2 - y_ctr = ymin + hs / 2 - valid_flag_x = paddle.less_equal(x_ctr, im_w) - valid_flag_y = paddle.less_equal(y_ctr, im_h) - valid_flag = paddle.logical_and(valid_flag, valid_flag_x) - valid_flag = paddle.logical_and(valid_flag, valid_flag_y) - - valid_flag = paddle.squeeze(valid_flag, axis=1) - valid_inds = paddle.nonzero(valid_flag) - return valid_inds - - def clip_tiled_boxes(self, im_w, im_h, input_boxes, pixel_offset=True): - offset = 1 if pixel_offset else 0 - xmin, ymin, xmax, ymax = paddle.tensor.split( - input_boxes, axis=1, num_or_sections=4) - xmin = paddle.clip(xmin, max=im_w - offset, min=0) - ymin = paddle.clip(ymin, max=im_h - offset, min=0) - xmax = paddle.clip(xmax, max=im_w - offset, min=0) - ymax = paddle.clip(ymax, max=im_h - offset, min=0) - input_boxes = paddle.concat([xmin, ymin, xmax, ymax], axis=1) - return input_boxes - - def box_encode(self, anchors, bbox_deltas, variances, pixel_offset=True): - offset = 1 if pixel_offset else 0 - anchor_xmin, anchor_ymin, anchor_xmax, anchor_ymax = paddle.tensor.split( - anchors, axis=1, num_or_sections=4) - anchor_width = anchor_xmax - anchor_xmin + offset - anchor_height = anchor_ymax - anchor_ymin + offset - anchor_center_x = anchor_xmin + 0.5 * anchor_width - anchor_center_y = anchor_ymin + 0.5 * anchor_height - var_center_x, var_center_y, var_width, var_height = paddle.tensor.split( - variances, axis=1, num_or_sections=4) - delta_center_x, delta_center_y, delta_width, delta_height = paddle.tensor.split( - bbox_deltas, axis=1, num_or_sections=4) - - bbox_center_x = var_center_x * delta_center_x * anchor_width + anchor_center_x - bbox_center_y = var_center_y * delta_center_y * anchor_height + anchor_center_y - bbox_width = paddle.exp( - paddle.clip( - var_width * delta_width, max=BBOX_CLIP_DEFAULT)) * anchor_width - bbox_height = paddle.exp( - paddle.clip( - var_height * delta_height, - max=BBOX_CLIP_DEFAULT)) * anchor_height - - proposal_xmin = bbox_center_x - bbox_width / 2 - proposal_ymin = bbox_center_y - bbox_height / 2 - proposal_xmax = bbox_center_x + bbox_width / 2 - offset - proposal_ymax = bbox_center_y + bbox_height / 2 - offset - proposal = paddle.concat( - [proposal_xmin, proposal_ymin, proposal_xmax, proposal_ymax], - axis=1) - return proposal - - def proposal_for_single_sample(self, anchors, bbox_deltas, im_info, scores, - variances): - proposal_num = paddle.shape(scores)[0] - pre_nms_top_n_tensor = paddle.assign( - np.asarray( - [self.pre_nms_topN], dtype='int32')) - k_candidate = paddle.concat([proposal_num, pre_nms_top_n_tensor]) - k = paddle.min(k_candidate) - scores, index = paddle.topk(scores, k=k, axis=0) - bbox_deltas = paddle.gather(bbox_deltas, index, axis=0) - anchors = paddle.gather(anchors, index, axis=0) - variances = paddle.gather(variances, index, axis=0) - - proposal = self.box_encode(anchors, bbox_deltas, variances, - self.pixel_offset) - if self.type == "generate_proposals_v2": - im_h, im_w = paddle.tensor.split(im_info, axis=1, num_or_sections=2) - else: - im_h, im_w, im_s = paddle.tensor.split( - im_info, axis=1, num_or_sections=3) - proposal = self.clip_tiled_boxes(im_w, im_h, proposal, - self.pixel_offset) - - if self.type == "generate_proposals_v2": - keep = self.filter_boxes_v2(proposal, im_w, im_h, self.min_size, - self.pixel_offset) - else: - keep = self.filter_boxes(proposal, im_w, im_h, im_s, self.min_size) - - tail_proposal = paddle.zeros(shape=[1, 4], dtype=proposal.dtype) - proposal_num = paddle.shape(proposal)[0] - tail_keep = paddle.reshape(proposal_num, shape=[1, 1]) - tail_keep = paddle.cast(tail_keep, dtype=keep.dtype) - tail_scores = paddle.zeros(shape=[1, 1], dtype=scores.dtype) - # proposal = paddle.concat([proposal, tail_proposal]) - # keep = paddle.concat([keep, tail_keep]) - # scores = paddle.concat([scores, tail_scores]) - - bbox_sel = paddle.gather(proposal, keep, axis=0) - scores_sel = paddle.gather(scores, keep, axis=0) - proposal = paddle.unsqueeze(bbox_sel, axis=0) - scores = paddle.transpose(scores_sel, perm=[1, 0]) - scores = paddle.unsqueeze(scores, axis=0) - out = layers.multiclass_nms( - proposal, - scores, - background_label=-1, - nms_top_k=self.pre_nms_topN, - score_threshold=-10000., - keep_top_k=self.post_nms_topN, - nms_threshold=self.nms_thresh, - normalized=False if self.pixel_offset else True, - nms_eta=self.eta) - label, scores, proposal = paddle.tensor.split( - out, axis=1, num_or_sections=[1, 1, 4]) - return scores, proposal - - def forward(self): - anchors = self.input('Anchors', 0) - bboxdeltas = self.input('BboxDeltas', 0) - if self.type == 'generate_proposals_v2': - iminfo = self.input('ImShape', 0) - else: - iminfo = self.input('ImInfo', 0) - scores = self.input('Scores', 0) - variances = self.input('Variances', 0) - - bboxdeltas = paddle.transpose(bboxdeltas, perm=[0, 2, 3, 1]) - bboxdeltas = paddle.reshape(bboxdeltas, [-1, 4]) - scores = paddle.transpose(scores, perm=[0, 2, 3, 1]) - scores = paddle.reshape(scores, [-1, 1]) - anchors = paddle.reshape(anchors, [-1, 4]) - variances = paddle.reshape(variances, [-1, 4]) - - new_scores, proposals = self.proposal_for_single_sample( - anchors, bboxdeltas, iminfo, scores, variances) - if len(self.node.outputs) == 3: - rois_num = paddle.shape(new_scores)[0] - return { - 'RpnRoiProbs': [new_scores], - 'RpnRois': [proposals], - 'RpnRoisNum': [rois_num] - } - else: - return {'RpnRoiProbs': [new_scores], 'RpnRois': [proposals]} - - -register_custom_paddle_op('generate_proposals', GenerateProposals) -register_custom_paddle_op('generate_proposals_v2', GenerateProposals) diff --git a/paddle2onnx/legacy/op_mapper/custom_paddle_op/grid_sampler.py b/paddle2onnx/legacy/op_mapper/custom_paddle_op/grid_sampler.py deleted file mode 100755 index 4191cbb53bd..00000000000 --- a/paddle2onnx/legacy/op_mapper/custom_paddle_op/grid_sampler.py +++ /dev/null @@ -1,151 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import numpy as np -import paddle -from paddle.fluid import layers -from paddle2onnx.legacy.op_mapper import CustomPaddleOp, register_custom_paddle_op -from paddle2onnx.legacy.op_mapper import OpMapper as op_mapper -from paddle2onnx.legacy.op_mapper import mapper_helper - - -class GridSampler(CustomPaddleOp): - def __init__(self, node, **kw): - super(GridSampler, self).__init__(node) - self.padding_mode = node.attr('padding_mode') - self.mode = node.attr('mode') - self.align_corners = node.attr('align_corners') - - def paddle_bilinear_grid_sample(self, im, grid, align_corners=False): - # this code reference: https://mmcv.readthedocs.io/en/latest/_modules/mmcv/ops/point_sample.html - im_shape = paddle.shape(im) - n, c, h, w = paddle.split(im_shape, num_or_sections=4) - grid_shape = paddle.shape(grid) - gn, gh, gw, _ = paddle.split(grid_shape, num_or_sections=4) - - # n, c, h, w = im.shape - # gn, gh, gw, _ = grid.shape - # assert n == gn - - x = grid[:, :, :, 0] - y = grid[:, :, :, 1] - - if align_corners: - x = ((x + 1) / 2) * (w - 1) - y = ((y + 1) / 2) * (h - 1) - else: - x = ((x + 1) * w - 1) / 2 - y = ((y + 1) * h - 1) / 2 - - x = paddle.reshape(x, [n, -1]) - y = paddle.reshape(y, [n, -1]) - - x0 = paddle.floor(x).astype('int64') - y0 = paddle.floor(y).astype('int64') - x1 = x0 + 1 - y1 = y0 + 1 - - x1_cast = x1.astype(grid.dtype) - x0_cast = x0.astype(grid.dtype) - y1_cast = y1.astype(grid.dtype) - y0_cast = y0.astype(grid.dtype) - wa = paddle.unsqueeze(((x1_cast - x) * (y1_cast - y)), 1) - wb = paddle.unsqueeze(((x1_cast - x) * (y - y0_cast)), 1) - wc = paddle.unsqueeze(((x - x0_cast) * (y1_cast - y)), 1) - wd = paddle.unsqueeze(((x - x0_cast) * (y - y0_cast)), 1) - - # Apply default for grid_sample function zero padding - im_padded = paddle.nn.functional.pad(im, - pad=[1, 1, 1, 1], - mode='constant', - value=0) - if im_padded.dtype != im.dtype: - im_padded = paddle.cast(im_padded, im.dtype) - padded_h = h + 2 - padded_w = w + 2 - # save points positions after padding - x0, x1, y0, y1 = x0 + 1, x1 + 1, y0 + 1, y1 + 1 - - # Clip coordinates to padded image size - tensor_zero = paddle.full(shape=[1], dtype='int64', fill_value=0.0) - tensor_padded_w = paddle.full( - shape=[1], dtype='int64', fill_value=padded_w - 1) - tensor_padded_h = paddle.full( - shape=[1], dtype='int64', fill_value=padded_h - 1) - x0 = paddle.where(x0 < 0, tensor_zero, x0) - x0 = paddle.where(x0 > padded_w - 1, tensor_padded_w, x0) - x1 = paddle.where(x1 < 0, tensor_zero, x1) - x1 = paddle.where(x1 > padded_w - 1, tensor_padded_w, x1) - y0 = paddle.where(y0 < 0, tensor_zero, y0) - y0 = paddle.where(y0 > padded_h - 1, tensor_padded_h, y0) - y1 = paddle.where(y1 < 0, tensor_zero, y1) - y1 = paddle.where(y1 > padded_h - 1, tensor_padded_h, y1) - im_padded = paddle.reshape(im_padded, [n, c, -1]) - - x0_y0 = paddle.expand( - paddle.unsqueeze((x0 + y0 * padded_w), 1), [-1, c, -1]) - x0_y1 = paddle.expand( - paddle.unsqueeze((x0 + y1 * padded_w), 1), [-1, c, -1]) - x1_y0 = paddle.expand( - paddle.unsqueeze((x1 + y0 * padded_w), 1), [-1, c, -1]) - x1_y1 = paddle.expand( - paddle.unsqueeze((x1 + y1 * padded_w), 1), [-1, c, -1]) - - Ia = self.paddle_gather(im_padded, 2, x0_y0) - Ib = self.paddle_gather(im_padded, 2, x0_y1) - Ic = self.paddle_gather(im_padded, 2, x1_y0) - Id = self.paddle_gather(im_padded, 2, x1_y1) - - return paddle.reshape((Ia * wa + Ib * wb + Ic * wc + Id * wd), - [n, c, gh, gw]) - - def paddle_gather(self, x, dim, index): - # index_shape = index.shape - index_shape = paddle.shape(index) - x_shape = paddle.shape(x) - index_flatten = index.flatten() - if dim < 0: - dim = len(x.shape) + dim - nd_index = [] - for k in range(len(x.shape)): - if k == dim: - nd_index.append(index_flatten) - else: - reshape_shape = [1] * len(x.shape) - x_shape_k = x_shape[k] - # x_shape_k = x.shape[k] - reshape_shape[k] = x_shape_k - x_arange = paddle.arange(x_shape_k, dtype=index.dtype) - x_arange = x_arange.reshape(reshape_shape) - dim_index = paddle.expand(x_arange, index_shape).flatten() - nd_index.append(dim_index) - ind2 = paddle.transpose(paddle.stack(nd_index), [1, 0]).astype("int64") - paddle_out = paddle.gather_nd(x, ind2).reshape(index_shape) - return paddle_out - - def forward(self): - input = self.input('X', 0) - grid = self.input('Grid', 0) - if self.mode != 'bilinear' or self.padding_mode != 'zeros': - raise Exception( - "grid_sample only is supported with mode should be 'bilinear' and padding_mode should be 'zeros'" - ) - res = self.paddle_bilinear_grid_sample( - input, grid, align_corners=self.align_corners) - return {'Output': [res]} - - -register_custom_paddle_op('grid_sampler', GridSampler) diff --git a/paddle2onnx/legacy/op_mapper/detection/__init__.py b/paddle2onnx/legacy/op_mapper/detection/__init__.py deleted file mode 100644 index 847ddc47ac8..00000000000 --- a/paddle2onnx/legacy/op_mapper/detection/__init__.py +++ /dev/null @@ -1,13 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. diff --git a/paddle2onnx/legacy/op_mapper/detection/box_coder.py b/paddle2onnx/legacy/op_mapper/detection/box_coder.py deleted file mode 100755 index 6e8df722bcb..00000000000 --- a/paddle2onnx/legacy/op_mapper/detection/box_coder.py +++ /dev/null @@ -1,363 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import numpy as np -from paddle2onnx.legacy.constant import dtypes -from paddle2onnx.legacy.op_mapper import OpMapper as op_mapper -from paddle2onnx.legacy.op_mapper import mapper_helper - - -@op_mapper('box_coder') -class BoxCoder(): - """ - we use the decode the prior box to target box, - we just use the decode mode to transform this op. - """ - support_opset_verison_range = (7, 12) - - @classmethod - def opset_7(cls, graph, node, **kw): - input_names = node.input_names - - t_size = node.input_shape('TargetBox', 0) - p_size = node.input_shape('PriorBox', 0) - - # get the outout_name - result_name = node.output('OutputBox', 0) - # n is size of batch, m is boxes num of targe_boxes - n = t_size[0] - m = t_size[0] - - axis = int(node.attr('axis')) - - #norm - norm = bool(node.attr('box_normalized')) - - name_slice_x1 = node.output('OutputBox')[0] + "@x1" - name_slice_y1 = node.output('OutputBox')[0] + "@y1" - name_slice_x2 = node.output('OutputBox')[0] + "@x2" - name_slice_y2 = node.output('OutputBox')[0] + "@y2" - - #make onnx tensor to save the intermeidate reslut - name_slice_indices = [ - [node.output('OutputBox')[0] + "@slice_" + str(i)] - for i in range(1, 3) - ] - node_slice_indices = [None for i in range(1, 3)] - - # create the range(0, 4) const data to slice - for i in range(1, 3): - tmp_node = graph.make_node( - 'Constant', - inputs=[], - outputs=name_slice_indices[i - 1], - dtype=dtypes.ONNX.FLOAT, - dims=(), - value=[i]) - # make node split data - name_box_split = [ - name_slice_x1, name_slice_y1, name_slice_x2, name_slice_y2 - ] - split_shape = list(p_size) - split_shape[-1] = 1 - - node_split_prior_node = graph.make_node( - 'Split', - inputs=node.input('PriorBox'), - outputs=name_box_split, - axis=1) - - # make node get centor node for decode - final_outputs_vars = [] - if not norm: - name_centor_w_tmp = [node.output('OutputBox')[0] + "@centor_w_tmp"] - name_centor_h_tmp = [node.output('OutputBox')[0] + "@centor_h_tmp"] - node_centor_w_tmp = None - node_centor_h_tmp = None - name_centor_tmp_list = [name_centor_w_tmp, name_centor_h_tmp] - node_centor_tmp_list = [node_centor_w_tmp, node_centor_h_tmp] - - count = 2 - for (name, op_node) in zip(name_centor_tmp_list, - node_centor_tmp_list): - tmp_node = graph.make_node('Add', - inputs=[node.output('OutputBox')[0] + "@slice_" + str(1)]\ - + [name_box_split[count]], - outputs=name) - count = count + 1 - if not norm: - inputs_sub = [[name_centor_w_tmp[0], name_box_split[0]], - [name_centor_h_tmp[0], name_box_split[1]]] - else: - inputs_sub = [[name_box_split[2], name_box_split[0]], - [name_box_split[3], name_box_split[1]]] - outputs_sub = [result_name + "@pb_w", result_name + "@pb_h"] - for i in range(0, 2): - tmp_node = graph.make_node( - 'Sub', inputs=inputs_sub[i], outputs=[outputs_sub[i]]) - # according to prior_box height and weight to get centor x, y - name_half_value = [result_name + "@half_value"] - node_half_value = graph.make_node( - 'Constant', - inputs=[], - outputs=name_half_value, - dtype=dtypes.ONNX.FLOAT, - dims=(), - value=[0.5]) - outputs_half_wh = [[result_name + "@pb_w_half"], - [result_name + "@pb_h_half"]] - inputs_half_wh = [[result_name + "@pb_w", name_half_value[0]], - [result_name + "@pb_h", name_half_value[0]]] - - for i in range(0, 2): - tmp_node = graph.make_node( - 'Mul', inputs=inputs_half_wh[i], outputs=outputs_half_wh[i]) - - inputs_centor_xy = [[outputs_half_wh[0][0], name_slice_x1], - [outputs_half_wh[1][0], name_slice_y1]] - - outputs_centor_xy = [[result_name + "@pb_x"], [result_name + "@pb_y"]] - - # final calc the centor x ,y - for i in range(0, 2): - tmp_node = graph.make_node( - 'Add', inputs=inputs_centor_xy[i], outputs=outputs_centor_xy[i]) - # reshape the data - shape = (1, split_shape[0]) if axis == 0 else (split_shape[0], 1) - - # need to reshape the data - inputs_transpose_pb = [ - [result_name + "@pb_w"], - [result_name + "@pb_h"], - [result_name + "@pb_x"], - [result_name + "@pb_y"], - ] - outputs_transpose_pb = [ - [result_name + "@pb_w_transpose"], - [result_name + "@pb_h_transpose"], - [result_name + "@pb_x_transpose"], - [result_name + "@pb_y_transpose"], - ] - if axis == 0: - name_reshape_pb = [result_name + "@pb_transpose"] - # reshape the data - for i in range(0, 4): - tmp_node = graph.make_node( - 'Transpose', - inputs=inputs_transpose_pb[i], - outputs=outputs_transpose_pb[i]) - # decoder the box according to the target_box and variacne - name_variance_raw = [result_name + "@variance_raw"] - name_variance_unsqueeze = [result_name + "@variance_unsqueeze"] - shape = [] - # make node to extend the data - var_split_axis = 0 - var_split_inputs_name = [] - if 'PriorBoxVar' in input_names and len(node.input('PriorBoxVar')) > 0: - if axis == 1: - raise Exception( - "The op box_coder has variable do not support aixs broadcast" - ) - axes = [] - var_split_inputs_name = [result_name + "@variance_split"] - tmp_node = graph.make_node( - 'Transpose', - inputs=node.input('PriorBoxVar'), - outputs=var_split_inputs_name) - var_split_axis = 0 - else: - variances = [1.0, 1.0, 1.0, 1.0] - if 'variance' in node.attrs and len(node.attr('variance')) > 0: - variances = [float(var) for var in node.attr('variance')] - node_variance_create = graph.make_node( - 'Constant', - inputs=[], - outputs=name_variance_raw, - dtype=dtypes.ONNX.FLOAT, - dims=[len(variances)], - value=variances) - var_split_axis = 0 - var_split_inputs_name = name_variance_raw - - # decode the result - outputs_split_variance = [ - result_name + "@variance_split" + str(i) for i in range(0, 4) - ] - outputs_split_targebox = [ - result_name + "@targebox_split" + str(i) for i in range(0, 4) - ] - node_split_var = graph.make_node( - 'Split', - inputs=var_split_inputs_name, - outputs=outputs_split_variance, - axis=var_split_axis) - node_split_target = graph.make_node( - 'Split', - inputs=node.input('TargetBox'), - outputs=outputs_split_targebox, - axis=2) - - outputs_squeeze_targebox = [ - result_name + "@targebox_squeeze" + str(i) for i in range(0, 4) - ] - for (input_name, output_name) in zip(outputs_split_targebox, - outputs_squeeze_targebox): - tmp_node = mapper_helper.squeeze_helper(graph, input_name, [2], - [output_name]) - - output_shape_step1 = list(t_size)[:-1] - - inputs_tb_step1 = [ - [outputs_squeeze_targebox[0], outputs_split_variance[0]], - [outputs_squeeze_targebox[1], outputs_split_variance[1]], - [outputs_squeeze_targebox[2], outputs_split_variance[2]], - [outputs_squeeze_targebox[3], outputs_split_variance[3]] - ] - outputs_tb_step1 = [[result_name + "@decode_x_step1"], - [result_name + "@decode_y_step1"], - [result_name + "@decode_w_step1"], - [result_name + "@decode_h_step1"]] - - for input_step1, output_step_1 in zip(inputs_tb_step1, - outputs_tb_step1): - tmp_node = graph.make_node( - 'Mul', inputs=input_step1, outputs=output_step_1) - if axis == 0: - inputs_tbxy_step2 = [[ - outputs_tb_step1[0][0], outputs_transpose_pb[0][0] - ], [outputs_tb_step1[1][0], outputs_transpose_pb[1][0]]] - else: - inputs_tbxy_step2 = [[ - outputs_tb_step1[0][0], inputs_transpose_pb[0][0] - ], [outputs_tb_step1[1][0], inputs_transpose_pb[1][0]]] - - outputs_tbxy_step2 = [[result_name + "@decode_x_step2"], - [result_name + "@decode_y_step2"]] - - for input_step2, output_step_2 in zip(inputs_tbxy_step2, - outputs_tbxy_step2): - tmp_node = graph.make_node( - 'Mul', inputs=input_step2, outputs=output_step_2) - if axis == 0: - inputs_tbxy_step3 = [[ - outputs_tbxy_step2[0][0], outputs_transpose_pb[2][0] - ], [outputs_tbxy_step2[1][0], outputs_transpose_pb[3][0]]] - else: - inputs_tbxy_step3 = [[ - outputs_tbxy_step2[0][0], inputs_transpose_pb[2][0] - ], [outputs_tbxy_step2[1][0], inputs_transpose_pb[3][0]]] - - outputs_tbxy_step3 = [[result_name + "@decode_x_step3"], - [result_name + "@decode_y_step3"]] - - for input_step3, output_step_3 in zip(inputs_tbxy_step3, - outputs_tbxy_step3): - tmp_node = graph.make_node( - 'Add', inputs=input_step3, outputs=output_step_3) - - # deal with width & height - inputs_tbwh_step2 = [outputs_tb_step1[2], outputs_tb_step1[3]] - outputs_tbwh_step2 = [[result_name + "@decode_w_step2"], - [result_name + "@decode_h_step2"]] - - for input_name, output_name in zip(inputs_tbwh_step2, - outputs_tbwh_step2): - tmp_node = graph.make_node( - 'Exp', inputs=input_name, outputs=output_name) - - if axis == 0: - inputs_tbwh_step3 = [[ - outputs_tbwh_step2[0][0], outputs_transpose_pb[0][0] - ], [outputs_tbwh_step2[1][0], outputs_transpose_pb[1][0]]] - else: - inputs_tbwh_step3 = [[ - outputs_tbwh_step2[0][0], inputs_transpose_pb[0][0] - ], [outputs_tbwh_step2[1][0], inputs_transpose_pb[1][0]]] - - outputs_tbwh_step3 = [[result_name + "@decode_w_step3"], - [result_name + "@decode_h_step3"]] - - for input_name, output_name in zip(inputs_tbwh_step3, - outputs_tbwh_step3): - tmp_node = graph.make_node( - 'Mul', inputs=input_name, outputs=output_name) - - # final step to calc the result, and concat the result to output - # return the output box, [(x1, y1), (x2, y2)] - - inputs_half_tbwh_step4 = [[ - outputs_tbwh_step3[0][0], result_name + "@slice_2" - ], [outputs_tbwh_step3[1][0], result_name + "@slice_2"]] - - outputs_half_tbwh_step4 = [[result_name + "@decode_half_w_step4"], - [result_name + "@decode_half_h_step4"]] - for inputs_name, outputs_name in zip(inputs_half_tbwh_step4, - outputs_half_tbwh_step4): - tmp_node = graph.make_node( - 'Div', inputs=inputs_name, outputs=outputs_name) - inputs_output_point1 = [[ - outputs_tbxy_step3[0][0], outputs_half_tbwh_step4[0][0] - ], [outputs_tbxy_step3[1][0], outputs_half_tbwh_step4[1][0]]] - - outputs_output_point1 = [[result_name + "@ouput_x1"], - [result_name + "@output_y1"]] - for input_name, output_name in zip(inputs_output_point1, - outputs_output_point1): - tmp_node = graph.make_node( - 'Sub', inputs=input_name, outputs=output_name) - - inputs_output_point2 = [[ - outputs_tbxy_step3[0][0], outputs_half_tbwh_step4[0][0] - ], [outputs_tbxy_step3[1][0], outputs_half_tbwh_step4[1][0]]] - - outputs_output_point2 = [[result_name + "@ouput_x2"], - [result_name + "@output_y2"]] - - for input_name, output_name in zip(inputs_output_point2, - outputs_output_point2): - tmp_node = graph.make_node( - 'Add', inputs=input_name, outputs=output_name) - if not norm: - inputs_unnorm_point2 = [[ - outputs_output_point2[0][0], result_name + "@slice_1" - ], [outputs_output_point2[1][0], result_name + "@slice_1"]] - outputs_unnorm_point2 = [[result_name + "@ouput_unnorm_x2"], - [result_name + "@ouput_unnorm_y2"]] - - for input_name, output_name in zip(inputs_unnorm_point2, - outputs_unnorm_point2): - tmp_node = graph.make_node( - 'Sub', inputs=input_name, outputs=output_name) - outputs_output_point2 = outputs_unnorm_point2 - - outputs_output_point1.extend(outputs_output_point2) - ouputs_points_unsqueeze = [[result_name + "@points_unsqueeze_x1"], - [result_name + "points_unsqueeze_y1"], - [result_name + "points_unsqueeze_x2"], - [result_name + "points_unsqueeze_y2"]] - - for input_name, output_name in zip(outputs_output_point1, - ouputs_points_unsqueeze): - tmp_node = mapper_helper.unsqueeze_helper( - graph, input_name, [len(output_shape_step1)], output_name) - outputs_points_unsqueeze_list = [ - output[0] for output in ouputs_points_unsqueeze - ] - node_point_final = graph.make_node( - 'Concat', - inputs=outputs_points_unsqueeze_list, - outputs=node.output('OutputBox'), - axis=len(output_shape_step1)) diff --git a/paddle2onnx/legacy/op_mapper/detection/density_prior_box.py b/paddle2onnx/legacy/op_mapper/detection/density_prior_box.py deleted file mode 100644 index f9ecc61b116..00000000000 --- a/paddle2onnx/legacy/op_mapper/detection/density_prior_box.py +++ /dev/null @@ -1,125 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import math -import numpy as np -from paddle2onnx.legacy.constant import dtypes -from paddle2onnx.legacy.op_mapper import OpMapper as op_mapper -from paddle2onnx.utils import require_fixed_shape - - -@op_mapper('density_prior_box') -class DensityPriorBox(): - """ - In this function, use the attribute to get the prior box, because we do not use - the image data and feature map, wo could the python code to create the varaible, - and to create the onnx tensor as output. - """ - support_opset_verison_range = (1, 12) - - @classmethod - def opset_9(cls, graph, node, **kw): - clip = bool(node.attr('clip')) - densities = node.attr('densities') - fixed_ratios = node.attr('fixed_ratios') - fixed_sizes = node.attr('fixed_sizes') - flatten_to_2d = bool(node.attr('flatten_to_2d')) - offset = node.attr('offset') - step_h = node.attr('step_h') - step_w = node.attr('step_w') - variances = node.attr('variances') - - input_shape = node.input_shape('Input', 0) - image_shape = node.input_shape('Image', 0) - - img_width = image_shape[3] - img_height = image_shape[2] - feature_width = input_shape[3] - feature_height = input_shape[2] - - assert img_width > 0 and img_height > 0, require_fixed_shape( - cls.__name__) - - if step_w == 0.0 or step_h == 0.0: - step_w = float(img_width / feature_width) - step_h = float(img_height / feature_height) - - num_priors = 0 - if len(fixed_sizes) > 0 and len(densities) > 0: - for density in densities: - if len(fixed_ratios) > 0: - num_priors += len(fixed_ratios) * (pow(density, 2)) - - out_dim = (feature_height, feature_width, num_priors, 4) - out_boxes = np.zeros(out_dim).astype('float32') - out_var = np.zeros(out_dim).astype('float32') - step_average = int((step_w + step_h) * 0.5) - - for h in range(feature_height): - for w in range(feature_width): - c_x = (w + offset) * step_w - c_y = (h + offset) * step_h - idx = 0 - - for density, fixed_size in zip(densities, fixed_sizes): - if (len(fixed_ratios) > 0): - for ar in fixed_ratios: - shift = int(step_average / density) - box_width_ratio = fixed_size * math.sqrt(ar) - box_height_ratio = fixed_size / math.sqrt(ar) - for di in range(density): - for dj in range(density): - c_x_temp = c_x - step_average / 2.0 + shift / 2.0 + dj * shift - c_y_temp = c_y - step_average / 2.0 + shift / 2.0 + di * shift - out_boxes[h, w, idx, :] = [ - max((c_x_temp - box_width_ratio / 2.0) / - img_width, 0), - max((c_y_temp - box_height_ratio / 2.0) - / img_height, 0), - min((c_x_temp + box_width_ratio / 2.0) / - img_width, 1), - min((c_y_temp + box_height_ratio / 2.0) - / img_height, 1) - ] - idx += 1 - - if clip: - out_boxes = np.clip(out_boxes, 0.0, 1.0) - # set the variance. - out_var = np.tile(variances, - (feature_height, feature_width, num_priors, 1)) - - if flatten_to_2d: - out_boxes = out_boxes.reshape((-1, 4)) - out_var = out_var.reshape((-1, 4)) - - #make node that - - node_boxes = graph.make_node( - 'Constant', - inputs=[], - outputs=node.output('Boxes'), - dtype=dtypes.ONNX.FLOAT, - dims=out_boxes.shape, - value=out_boxes.flatten().tolist()) - - node_vars = graph.make_node( - 'Constant', - inputs=[], - outputs=node.output('Variances'), - dtype=dtypes.ONNX.FLOAT, - dims=out_var.shape, - value=out_var.flatten().tolist()) diff --git a/paddle2onnx/legacy/op_mapper/detection/multiclass_nms.py b/paddle2onnx/legacy/op_mapper/detection/multiclass_nms.py deleted file mode 100755 index 9e57ff8fc54..00000000000 --- a/paddle2onnx/legacy/op_mapper/detection/multiclass_nms.py +++ /dev/null @@ -1,343 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -import numpy as np -from paddle2onnx.utils import logging -from paddle2onnx.legacy.constant import dtypes -from paddle2onnx.legacy.op_mapper import OpMapper as op_mapper -from paddle2onnx.legacy.op_mapper import mapper_helper - - -@op_mapper( - ['multiclass_nms', 'multiclass_nms2', 'matrix_nms', 'multiclass_nms3']) -class MultiClassNMS(): - support_opset_verision_range = (10, 16) - """ - Convert the paddle multiclass_nms to onnx op. - This op is get the select boxes from origin boxes. - """ - - @classmethod - def opset_10(cls, graph, node, **kw): - if node.input_shape("BBoxes", 0)[0] != 1: - logging.warning( - "Due to the operator:{}, the converted ONNX model will only supports input[batch_size] == 1.". - format(node.type)) - scores = node.input('Scores', 0) - bboxes = node.input('BBoxes', 0) - num_class = node.input_shape('Scores', 0)[1] - if len(node.input_shape('Scores', 0)) == 2: - # inputs: scores & bboxes is lod tensor - scores = graph.make_node('Transpose', inputs=[scores], perm=[1, 0]) - scores = mapper_helper.unsqueeze_helper(graph, scores, [0]) - if graph.opset_version < 13: - scores_list = graph.make_node( - 'Split', - inputs=scores, - outputs=num_class, - axis=1, - split=[1] * num_class) - else: - split_const = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[1] * num_class) - scores_list = graph.make_node( - "Split", - inputs=[scores] + [split_const], - outputs=num_class, - axis=1) - - bboxes = graph.make_node('Transpose', inputs=bboxes, perm=[1, 0, 2]) - if graph.opset_version < 13: - bboxes_list = graph.make_node( - 'Split', - inputs=bboxes, - outputs=num_class, - axis=0, - split=[1] * num_class) - else: - split_const = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[1] * num_class) - bboxes_list = graph.make_node( - "Split", - inputs=[bboxes] + [split_const], - outputs=num_class, - axis=0) - bbox_ids = [] - if not isinstance(scores_list, list): - scores_list = [scores_list] - if not isinstance(bboxes_list, list): - bboxes_list = [bboxes_list] - for i in range(num_class): - bbox_id = cls.nms(graph, - node, - scores_list[i], - bboxes_list[i], - class_id=i) - bbox_ids.append(bbox_id) - bbox_ids = graph.make_node('Concat', inputs=bbox_ids, axis=0) - const_shape = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[1, -1, 4]) - bboxes = graph.make_node('Reshape', inputs=[bboxes, const_shape]) - cls.keep_top_k( - graph, node, bbox_ids, scores, bboxes, is_lod_input=True) - else: - bbox_ids = cls.nms(graph, node, scores, bboxes) - cls.keep_top_k(graph, node, bbox_ids, scores, bboxes) - - @classmethod - def nms(cls, graph, node, scores, bboxes, class_id=None): - normalized = node.attr('normalized') - nms_top_k = node.attr('nms_top_k') - if node.type == 'matrix_nms': - iou_threshold = 0.5 - logging.warning( - "Operator:{} is not supported completely, so we use traditional" - " NMS (nms_theshold={}) to instead it, which introduce some difference.". - format(node.type, str(iou_threshold))) - else: - iou_threshold = node.attr('nms_threshold') - if nms_top_k == -1: - nms_top_k = 100000 - - #convert the paddle attribute to onnx tensor - score_threshold = graph.make_node( - 'Constant', - dtype=dtypes.ONNX.FLOAT, - value=[float(node.attr('score_threshold'))]) - iou_threshold = graph.make_node( - 'Constant', dtype=dtypes.ONNX.FLOAT, value=[float(iou_threshold)]) - nms_top_k = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[np.int64(nms_top_k)]) - - # the paddle data format is x1,y1,x2,y2 - kwargs = {'center_point_box': 0} - - if normalized: - select_bbox_indices = graph.make_node( - 'NonMaxSuppression', - inputs=[ - bboxes, scores, nms_top_k, iou_threshold, score_threshold - ]) - elif not normalized: - value_one = graph.make_node( - 'Constant', dims=[1], dtype=dtypes.ONNX.FLOAT, value=1.0) - if graph.opset_version < 13: - new_bboxes = graph.make_node( - 'Split', - inputs=[bboxes], - outputs=4, - axis=2, - split=[1, 1, 1, 1]) - else: - split_const = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[1, 1, 1, 1]) - new_bboxes = graph.make_node( - "Split", inputs=[bboxes] + [split_const], outputs=4, axis=2) - new_xmax = graph.make_node('Add', inputs=[new_bboxes[2], value_one]) - new_ymax = graph.make_node('Add', inputs=[new_bboxes[3], value_one]) - new_bboxes = graph.make_node( - 'Concat', - inputs=[new_bboxes[0], new_bboxes[1], new_xmax, new_ymax], - axis=2) - select_bbox_indices = graph.make_node( - 'NonMaxSuppression', - inputs=[ - new_bboxes, scores, nms_top_k, iou_threshold, - score_threshold - ]) - - if class_id is not None and class_id != 0: - class_id = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[0, class_id, 0]) - class_id = mapper_helper.unsqueeze_helper(graph, class_id, [0]) - select_bbox_indices = graph.make_node( - 'Add', inputs=[select_bbox_indices, class_id]) - - return select_bbox_indices - - @classmethod - def keep_top_k(cls, - graph, - node, - select_bbox_indices, - scores, - bboxes, - is_lod_input=False): - # step 1 nodes select the nms class - # create some const value to use - background = node.attr('background_label') - const_values = [] - for value in [0, 1, 2, -1]: - const_value = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[value]) - const_values.append(const_value) - - # In this code block, we will deocde the raw score data, reshape N * C * M to 1 * N*C*M - # and the same time, decode the select indices to 1 * D, gather the select_indices - class_id = graph.make_node( - 'Gather', inputs=[select_bbox_indices, const_values[1]], axis=1) - - squeezed_class_id = mapper_helper.squeeze_helper(graph, class_id, [1]) - - bbox_id = graph.make_node( - 'Gather', inputs=[select_bbox_indices, const_values[2]], axis=1) - - if background == 0: - nonzero = graph.make_node('NonZero', inputs=[squeezed_class_id]) - else: - filter_cls_id = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT32, value=[background]) - cast = graph.make_node( - 'Cast', inputs=[squeezed_class_id], to=dtypes.ONNX.INT32) - filter_index = graph.make_node('Sub', inputs=[cast, filter_cls_id]) - nonzero = graph.make_node('NonZero', inputs=[filter_index]) - - class_id = graph.make_node('Gather', inputs=[class_id, nonzero], axis=0) - class_id = graph.make_node( - 'Cast', inputs=[class_id], to=dtypes.ONNX.INT64) - - bbox_id = graph.make_node('Gather', inputs=[bbox_id, nonzero], axis=0) - bbox_id = graph.make_node( - 'Cast', inputs=[bbox_id], to=dtypes.ONNX.INT64) - - # get the shape of scores - shape_scores = graph.make_node('Shape', inputs=scores) - - # gather the index: 2 shape of scores - class_num = graph.make_node( - 'Gather', inputs=[shape_scores, const_values[2]], axis=0) - - # reshape scores N * C * M to (N*C*M) * 1 - scores = graph.make_node('Reshape', inputs=[scores, const_values[-1]]) - - # mul class * M - mul_classnum_boxnum = graph.make_node( - 'Mul', inputs=[class_id, class_num]) - - # add class * M * index - add_class_indices = graph.make_node( - 'Add', inputs=[mul_classnum_boxnum, bbox_id]) - - # Squeeze the indices to 1 dim - score_indices = mapper_helper.squeeze_helper(graph, add_class_indices, - [0, 2]) - - # gather the data from flatten scores - scores = graph.make_node( - 'Gather', inputs=[scores, score_indices], axis=0) - - keep_top_k = node.attr('keep_top_k') - keep_top_k = graph.make_node( - 'Constant', - dtype=dtypes.ONNX.INT64, - dims=[1, 1], - value=[node.attr('keep_top_k')]) - - # get min(topK, num_select) - shape_select_num = graph.make_node('Shape', inputs=[scores]) - const_zero = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[0]) - gather_select_num = graph.make_node( - 'Gather', inputs=[shape_select_num, const_zero], axis=0) - unsqueeze_select_num = mapper_helper.unsqueeze_helper( - graph, gather_select_num, [0]) - - concat_topK_select_num = graph.make_node( - 'Concat', inputs=[unsqueeze_select_num, keep_top_k], axis=0) - cast_concat_topK_select_num = graph.make_node( - 'Cast', inputs=[concat_topK_select_num], to=6) - keep_top_k = graph.make_node( - 'ReduceMin', inputs=[cast_concat_topK_select_num], keepdims=0) - # unsqueeze the indices to 1D tensor - keep_top_k = mapper_helper.unsqueeze_helper(graph, keep_top_k, [0]) - - # cast the indices to INT64 - keep_top_k = graph.make_node('Cast', inputs=[keep_top_k], to=7) - - # select topk scores indices - keep_topk_scores, keep_topk_indices = graph.make_node( - 'TopK', inputs=[scores, keep_top_k], outputs=2) - - # gather topk label, scores, boxes - gather_topk_scores = graph.make_node( - 'Gather', inputs=[scores, keep_topk_indices], axis=0) - - gather_topk_class = graph.make_node( - 'Gather', inputs=[class_id, keep_topk_indices], axis=1) - - # gather the boxes need to gather the boxes id, then get boxes - if is_lod_input: - gather_topk_boxes_id = graph.make_node( - 'Gather', [add_class_indices, keep_topk_indices], axis=1) - else: - gather_topk_boxes_id = graph.make_node( - 'Gather', [bbox_id, keep_topk_indices], axis=1) - - # squeeze the gather_topk_boxes_id to 1 dim - squeeze_topk_boxes_id = mapper_helper.squeeze_helper( - graph, gather_topk_boxes_id, [0, 2]) - - gather_select_boxes = graph.make_node( - 'Gather', inputs=[bboxes, squeeze_topk_boxes_id], axis=1) - - # concat the final result - # before concat need to cast the class to float - cast_topk_class = graph.make_node( - 'Cast', inputs=[gather_topk_class], to=1) - - unsqueeze_topk_scores = mapper_helper.unsqueeze_helper( - graph, gather_topk_scores, [0, 2]) - - inputs_concat_final_results = [ - cast_topk_class, unsqueeze_topk_scores, gather_select_boxes - ] - - sort_by_socre_results = graph.make_node( - 'Concat', inputs=inputs_concat_final_results, axis=2) - - # sort by class_id - squeeze_cast_topk_class = mapper_helper.squeeze_helper( - graph, cast_topk_class, [0, 2]) - - neg_squeeze_cast_topk_class = graph.make_node( - 'Neg', inputs=[squeeze_cast_topk_class]) - - data, indices = graph.make_node( - 'TopK', inputs=[neg_squeeze_cast_topk_class, keep_top_k], outputs=2) - - concat_final_results = graph.make_node( - 'Gather', inputs=[sort_by_socre_results, indices], axis=1) - - concat_final_results = mapper_helper.squeeze_helper( - graph, concat_final_results, [0], node.output('Out')) - - if node.type in ['multiclass_nms2', 'matrix_nms', 'multiclass_nms3']: - final_indices = mapper_helper.squeeze_helper(graph, bbox_id, [0], - node.output('Index')) - if node.type in ['matrix_nms', 'multiclass_nms3']: - select_bboxes_shape = graph.make_node('Shape', inputs=[indices]) - select_bboxes_shape1 = graph.make_node( - 'Cast', inputs=[select_bboxes_shape], to=dtypes.ONNX.INT32) - indices = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[0]) - rois_num = None - if 'NmsRoisNum' in node.outputs: - rois_num = node.output('NmsRoisNum') - elif 'RoisNum' in node.outputs: - rois_num = node.output('RoisNum') - if rois_num is not None: - graph.make_node( - "Gather", - inputs=[select_bboxes_shape1, indices], - outputs=rois_num) diff --git a/paddle2onnx/legacy/op_mapper/detection/prior_box.py b/paddle2onnx/legacy/op_mapper/detection/prior_box.py deleted file mode 100644 index 82cb4045279..00000000000 --- a/paddle2onnx/legacy/op_mapper/detection/prior_box.py +++ /dev/null @@ -1,176 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import math -import numpy as np -from paddle2onnx.legacy.constant import dtypes -from paddle2onnx.legacy.op_mapper import OpMapper as op_mapper -from paddle2onnx.utils import require_fixed_shape - - -def expand_aspect_rations(input_aspect_ratior, flip): - expsilon = 1e-6 - output_ratios = [1.0] - for input_ratio in input_aspect_ratior: - already_exis = False - for output_ratio in output_ratios: - if abs(input_ratio - output_ratio) < expsilon: - already_exis = True - break - if already_exis == False: - output_ratios.append(input_ratio) - if flip: - output_ratios.append(1.0 / input_ratio) - return output_ratios - - -@op_mapper('prior_box') -class PriorBox(): - """ - In this function, use the attribute to get the prior box, because we do not use - the image data and feature map, wo could the python code to create the varaible, - and to create the onnx tensor as output. - """ - support_opset_verison_range = (1, 12) - - @classmethod - def opset_9(cls, graph, node, **kw): - flip = bool(node.attr('flip')) - clip = bool(node.attr('clip')) - min_max_aspect_ratios_order = bool( - node.attr('min_max_aspect_ratios_order')) - min_sizes = [float(size) for size in node.attr('min_sizes')] - max_sizes = [float(size) for size in node.attr('max_sizes')] - if isinstance(node.attr('aspect_ratios'), list): - aspect_ratios = [ - float(ratio) for ratio in node.attr('aspect_ratios') - ] - else: - aspect_ratios = [float(node.attr('aspect_ratios'))] - variances = [float(var) for var in node.attr('variances')] - # set min_max_aspect_ratios_order = false - output_ratios = expand_aspect_rations(aspect_ratios, flip) - - step_w = float(node.attr('step_w')) - step_h = float(node.attr('step_h')) - offset = float(node.attr('offset')) - - input_shape = node.input_shape('Input', 0) - image_shape = node.input_shape('Image', 0) - - img_width = image_shape[3] - img_height = image_shape[2] - feature_width = input_shape[3] - feature_height = input_shape[2] - assert img_width > 0 and img_height > 0, require_fixed_shape( - cls.__name__) - - step_width = 1.0 - step_height = 1.0 - - if step_w == 0.0 or step_h == 0.0: - step_w = float(img_width / feature_width) - step_h = float(img_height / feature_height) - - num_priors = len(output_ratios) * len(min_sizes) - if len(max_sizes) > 0: - num_priors += len(max_sizes) - out_dim = (feature_height, feature_width, num_priors, 4) - out_boxes = np.zeros(out_dim).astype('float32') - out_var = np.zeros(out_dim).astype('float32') - - idx = 0 - for h in range(feature_height): - for w in range(feature_width): - c_x = (w + offset) * step_w - c_y = (h + offset) * step_h - idx = 0 - for s in range(len(min_sizes)): - min_size = min_sizes[s] - if not min_max_aspect_ratios_order: - # rest of priors - for r in range(len(output_ratios)): - ar = output_ratios[r] - c_w = min_size * math.sqrt(ar) / 2 - c_h = (min_size / math.sqrt(ar)) / 2 - out_boxes[h, w, idx, :] = [(c_x - c_w) / img_width, - (c_y - c_h) / img_height, - (c_x + c_w) / img_width, - (c_y + c_h) / img_height] - idx += 1 - - if len(max_sizes) > 0: - max_size = max_sizes[s] - # second prior: aspect_ratio = 1, - c_w = c_h = math.sqrt(min_size * max_size) / 2 - out_boxes[h, w, idx, :] = [(c_x - c_w) / img_width, - (c_y - c_h) / img_height, - (c_x + c_w) / img_width, - (c_y + c_h) / img_height] - idx += 1 - else: - c_w = c_h = min_size / 2. - out_boxes[h, w, idx, :] = [ - (c_x - c_w) / img_width, (c_y - c_h) / img_height, - (c_x + c_w) / img_width, (c_y + c_h) / img_height - ] - idx += 1 - if len(max_sizes) > 0: - max_size = max_sizes[s] - # second prior: aspect_ratio = 1, - c_w = c_h = math.sqrt(min_size * max_size) / 2 - out_boxes[h, w, idx, :] = [(c_x - c_w) / img_width, - (c_y - c_h) / img_height, - (c_x + c_w) / img_width, - (c_y + c_h) / img_height] - idx += 1 - - # rest of priors - for r in range(len(output_ratios)): - ar = output_ratios[r] - if abs(ar - 1.) < 1e-6: - continue - c_w = min_size * math.sqrt(ar) / 2 - c_h = (min_size / math.sqrt(ar)) / 2 - out_boxes[h, w, idx, :] = [(c_x - c_w) / img_width, - (c_y - c_h) / img_height, - (c_x + c_w) / img_width, - (c_y + c_h) / img_height] - idx += 1 - - if clip: - out_boxes = np.clip(out_boxes, 0.0, 1.0) - # set the variance. - out_var = np.tile(variances, - (feature_height, feature_width, num_priors, 1)) - - #make node that - - node_boxes = graph.make_node( - 'Constant', - inputs=[], - outputs=node.output('Boxes'), - dtype=dtypes.ONNX.FLOAT, - dims=out_boxes.shape, - value=out_boxes.flatten().tolist()) - - node_vars = graph.make_node( - 'Constant', - inputs=[], - outputs=node.output('Variances'), - dtype=dtypes.ONNX.FLOAT, - dims=out_var.shape, - value=out_var.flatten().tolist()) diff --git a/paddle2onnx/legacy/op_mapper/detection/yolo_box.py b/paddle2onnx/legacy/op_mapper/detection/yolo_box.py deleted file mode 100755 index 023dabc81e4..00000000000 --- a/paddle2onnx/legacy/op_mapper/detection/yolo_box.py +++ /dev/null @@ -1,506 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import sys -import numpy as np -from paddle2onnx.legacy.constant import dtypes -from paddle2onnx.legacy.op_mapper import OpMapper as op_mapper -from paddle2onnx.legacy.op_mapper import mapper_helper -from onnx import TensorProto -import paddle - -MAX_FLOAT32 = 3.402823466E+38 - - -@op_mapper('yolo_box') -class YOLOBox(): - support_opset_verison_range = (9, 12) - - node_pred_box_x1_decode = None - node_pred_box_y1_decode = None - node_pred_box_x2_decode = None - node_pred_box_y2_decode = None - node_pred_box_x2_sub_w = None - node_pred_box_y2_sub_h = None - - @classmethod - def opset_9(cls, graph, node, **kw): - model_name = node.output('Boxes', 0) - input_shape = node.input_shape('X', 0) - mapper_helper.is_static_shape(input_shape) - image_size = node.input('ImgSize') - input_height = input_shape[2] - input_width = input_shape[3] - class_num = node.attr('class_num') - anchors = node.attr('anchors') - num_anchors = int(len(anchors)) // 2 - scale_x_y = node.attr('scale_x_y') - downsample_ratio = node.attr('downsample_ratio') - input_size = input_height * downsample_ratio - conf_thresh = node.attr('conf_thresh') - conf_thresh_mat = [conf_thresh - ] * num_anchors * input_height * input_width - - cls.score_shape = [ - 1, input_height * input_width * int(num_anchors), class_num - ] - - im_outputs = [] - - x_shape = [1, num_anchors, 5 + class_num, input_height, input_width] - node_x_shape = graph.make_node( - 'Constant', attrs={'dtype': dtypes.ONNX.INT64, - 'value': x_shape}) - - node_x_reshape = graph.make_node( - 'Reshape', inputs=[node.input('X')[0], node_x_shape]) - node_x_transpose = graph.make_node( - 'Transpose', inputs=[node_x_reshape], perm=[0, 1, 3, 4, 2]) - - range_x = [] - range_y = [] - for i in range(0, input_width): - range_x.append(i) - for j in range(0, input_height): - range_y.append(j) - - node_range_x = graph.make_node( - 'Constant', - attrs={ - 'dtype': dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)], - 'value': range_x, - }) - - node_range_y = graph.make_node( - 'Constant', - inputs=[], - attrs={ - 'dtype': dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)], - 'value': range_y, - }) - - range_x_new_shape = [1, input_width] - range_y_new_shape = [input_height, 1] - - node_range_x_new_shape = graph.make_node( - 'Constant', - inputs=[], - dtype=dtypes.ONNX.INT64, - value=range_x_new_shape) - node_range_y_new_shape = graph.make_node( - 'Constant', - inputs=[], - dtype=dtypes.ONNX.INT64, - value=range_y_new_shape) - - node_range_x_reshape = graph.make_node( - 'Reshape', inputs=[node_range_x, node_range_x_new_shape]) - node_range_y_reshape = graph.make_node( - 'Reshape', inputs=[node_range_y, node_range_y_new_shape]) - - node_grid_x = graph.make_node( - "Tile", inputs=[node_range_x_reshape, node_range_y_new_shape]) - - node_grid_y = graph.make_node( - "Tile", inputs=[node_range_y_reshape, node_range_x_new_shape]) - - node_box_x = model_name + "@box_x" - node_box_y = model_name + "@box_y" - node_box_w = model_name + "@box_w" - node_box_h = model_name + "@box_h" - node_conf = model_name + "@conf" - node_prob = model_name + "@prob" - output = [ - node_box_x, node_box_y, node_box_w, node_box_h, node_conf, node_prob - ] - node_split_input = mapper_helper.split_helper( - graph, [node_x_transpose], - output, - -1, [1, 1, 1, 1, 1, class_num], - dtype=node.input_dtype('X', 0)) - - node_box_x_sigmoid = graph.make_node("Sigmoid", inputs=[node_box_x]) - - node_box_y_sigmoid = graph.make_node("Sigmoid", inputs=[node_box_y]) - - if scale_x_y is not None: - bias_x_y = -0.5 * (scale_x_y - 1.0) - scale_x_y_node = graph.make_node( - 'Constant', - attrs={ - 'dtype': - dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)], - 'value': scale_x_y - }) - - bias_x_y_node = graph.make_node( - 'Constant', - attrs={ - 'dtype': - dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)], - 'value': bias_x_y - }) - node_box_x_sigmoid = graph.make_node( - "Mul", inputs=[node_box_x_sigmoid, scale_x_y_node]) - node_box_x_sigmoid = graph.make_node( - "Add", inputs=[node_box_x_sigmoid, bias_x_y_node]) - node_box_y_sigmoid = graph.make_node( - "Mul", inputs=[node_box_y_sigmoid, scale_x_y_node]) - node_box_y_sigmoid = graph.make_node( - "Add", inputs=[node_box_y_sigmoid, bias_x_y_node]) - node_box_x_squeeze = mapper_helper.squeeze_helper( - graph, node_box_x_sigmoid, [4]) - - node_box_y_squeeze = mapper_helper.squeeze_helper( - graph, node_box_y_sigmoid, [4]) - - node_box_x_add_grid = graph.make_node( - "Add", inputs=[node_grid_x, node_box_x_squeeze]) - - node_box_y_add_grid = graph.make_node( - "Add", inputs=[node_grid_y, node_box_y_squeeze]) - - node_input_h = graph.make_node( - 'Constant', - inputs=[], - dtype=dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)], - value=[input_height]) - - node_input_w = graph.make_node( - 'Constant', - inputs=[], - dtype=dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)], - value=[input_width]) - - node_box_x_encode = graph.make_node( - 'Div', inputs=[node_box_x_add_grid, node_input_w]) - - node_box_y_encode = graph.make_node( - 'Div', inputs=[node_box_y_add_grid, node_input_h]) - - node_anchor_tensor = graph.make_node( - "Constant", - inputs=[], - dtype=dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)], - value=anchors) - - anchor_shape = [int(num_anchors), 2] - node_anchor_shape = graph.make_node( - "Constant", inputs=[], dtype=dtypes.ONNX.INT64, value=anchor_shape) - - node_anchor_tensor_reshape = graph.make_node( - "Reshape", inputs=[node_anchor_tensor, node_anchor_shape]) - - node_input_size = graph.make_node( - "Constant", - inputs=[], - dtype=dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)], - value=[input_size]) - - node_anchors_div_input_size = graph.make_node( - "Div", inputs=[node_anchor_tensor_reshape, node_input_size]) - - node_anchor_w = model_name + "@anchor_w" - node_anchor_h = model_name + "@anchor_h" - - node_anchor_split = mapper_helper.split_helper( - graph, - inputs=node_anchors_div_input_size, - axis=1, - split=[1, 1], - outputs=[node_anchor_w, node_anchor_h], - dtype=node.input_dtype('X', 0)) - - new_anchor_shape = [1, int(num_anchors), 1, 1] - node_new_anchor_shape = graph.make_node( - 'Constant', - inputs=[], - dtype=dtypes.ONNX.INT64, - value=new_anchor_shape) - - node_anchor_w_reshape = graph.make_node( - 'Reshape', inputs=[node_anchor_w, node_new_anchor_shape]) - - node_anchor_h_reshape = graph.make_node( - 'Reshape', inputs=[node_anchor_h, node_new_anchor_shape]) - - node_box_w_squeeze = mapper_helper.squeeze_helper(graph, node_box_w, - [4]) - node_box_h_squeeze = mapper_helper.squeeze_helper(graph, node_box_h, - [4]) - - node_box_w_exp = graph.make_node("Exp", inputs=[node_box_w_squeeze]) - node_box_h_exp = graph.make_node("Exp", inputs=[node_box_h_squeeze]) - - node_box_w_encode = graph.make_node( - 'Mul', inputs=[node_box_w_exp, node_anchor_w_reshape]) - - node_box_h_encode = graph.make_node( - 'Mul', inputs=[node_box_h_exp, node_anchor_h_reshape]) - - node_conf_sigmoid = graph.make_node('Sigmoid', inputs=[node_conf]) - - node_conf_thresh = graph.make_node( - 'Constant', - inputs=[], - dtype=dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)], - value=conf_thresh_mat) - - conf_shape = [1, int(num_anchors), input_height, input_width, 1] - node_conf_shape = graph.make_node( - 'Constant', inputs=[], dtype=dtypes.ONNX.INT64, value=conf_shape) - - node_conf_thresh_reshape = graph.make_node( - 'Reshape', inputs=[node_conf_thresh, node_conf_shape]) - - node_conf_sub = graph.make_node( - 'Sub', inputs=[node_conf_sigmoid, node_conf_thresh_reshape]) - - node_conf_clip = mapper_helper.clip_helper(graph, node, node_conf_sub, - float(MAX_FLOAT32), 0.0) - - node_zeros = graph.make_node( - 'Constant', - inputs=[], - dtype=dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)], - value=[0]) - - node_conf_clip_bool = graph.make_node( - 'Greater', inputs=[node_conf_clip, node_zeros]) - - node_conf_clip_cast = graph.make_node( - 'Cast', - inputs=[node_conf_clip_bool], - to=dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)]) - - node_conf_set_zero = graph.make_node( - 'Mul', inputs=[node_conf_sigmoid, node_conf_clip_cast]) - - node_prob_sigmoid = graph.make_node('Sigmoid', inputs=[node_prob]) - - new_shape = [1, int(num_anchors), input_height, input_width, 1] - node_new_shape = graph.make_node( - 'Constant', - inputs=[], - dtype=dtypes.ONNX.INT64, - dims=[len(new_shape)], - value=new_shape) - - node_conf_new_shape = graph.make_node( - 'Reshape', inputs=[node_conf_set_zero, node_new_shape]) - - cls.node_score = graph.make_node( - 'Mul', inputs=[node_prob_sigmoid, node_conf_new_shape]) - - node_conf_bool = graph.make_node( - 'Greater', inputs=[node_conf_new_shape, node_zeros]) - - node_box_x_new_shape = graph.make_node( - 'Reshape', inputs=[node_box_x_encode, node_new_shape]) - - node_box_y_new_shape = graph.make_node( - 'Reshape', inputs=[node_box_y_encode, node_new_shape]) - - node_box_w_new_shape = graph.make_node( - 'Reshape', inputs=[node_box_w_encode, node_new_shape]) - - node_box_h_new_shape = graph.make_node( - 'Reshape', inputs=[node_box_h_encode, node_new_shape]) - - node_pred_box = graph.make_node( - 'Concat', - inputs=[node_box_x_new_shape, node_box_y_new_shape, \ - node_box_w_new_shape, node_box_h_new_shape], - axis=4) - - node_conf_cast = graph.make_node( - 'Cast', - inputs=[node_conf_bool], - to=dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)]) - - node_pred_box_mul_conf = graph.make_node( - 'Mul', inputs=[node_pred_box, node_conf_cast]) - - box_shape = [1, int(num_anchors) * input_height * input_width, 4] - node_box_shape = graph.make_node( - 'Constant', inputs=[], dtype=dtypes.ONNX.INT64, value=box_shape) - - node_pred_box_new_shape = graph.make_node( - 'Reshape', inputs=[node_pred_box_mul_conf, node_box_shape]) - - node_pred_box_x = model_name + "@_pred_box_x" - node_pred_box_y = model_name + "@_pred_box_y" - node_pred_box_w = model_name + "@_pred_box_w" - node_pred_box_h = model_name + "@_pred_box_h" - if node.input_dtype('X', 0) == paddle.float64: - node_pred_box_new_shape = graph.make_node( - 'Cast', inputs=[node_pred_box_new_shape], to=TensorProto.FLOAT) - node_pred_box_split = mapper_helper.split_helper( - graph, - inputs=node_pred_box_new_shape, - axis=2, - split=[1, 1, 1, 1], - outputs=[ - node_pred_box_x, node_pred_box_y, node_pred_box_w, - node_pred_box_h - ], - dtype=node.input_dtype('X', 0)) - - if node.input_dtype('X', 0) == paddle.float64: - node_pred_box_x = graph.make_node( - 'Cast', - inputs=[node_pred_box_x], - to=dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)]) - node_pred_box_y = graph.make_node( - 'Cast', - inputs=[node_pred_box_y], - to=dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)]) - node_pred_box_w = graph.make_node( - 'Cast', - inputs=[node_pred_box_w], - to=dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)]) - node_pred_box_h = graph.make_node( - 'Cast', - inputs=[node_pred_box_h], - to=dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)]) - node_number_two = graph.make_node( - "Constant", - inputs=[], - dtype=dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)], - value=[2]) - - node_half_w = graph.make_node( - "Div", inputs=[node_pred_box_w, node_number_two]) - - node_half_h = graph.make_node( - "Div", inputs=[node_pred_box_h, node_number_two]) - - node_pred_box_x1 = graph.make_node( - 'Sub', inputs=[node_pred_box_x, node_half_w]) - - node_pred_box_y1 = graph.make_node( - 'Sub', inputs=[node_pred_box_y, node_half_h]) - - node_pred_box_x2 = graph.make_node( - 'Add', inputs=[node_pred_box_x, node_half_w]) - - node_pred_box_y2 = graph.make_node( - 'Add', inputs=[node_pred_box_y, node_half_h]) - - node_sqeeze_image_size = mapper_helper.squeeze_helper( - graph, image_size[0], [0]) - - node_img_height = model_name + "@img_height" - node_img_width = model_name + "@img_width" - node_image_size_split = mapper_helper.split_helper( - graph, [node_sqeeze_image_size], [node_img_height, node_img_width], - -1, [1, 1], - dtype=node.input_dtype('X', 0)) - - node_img_width_cast = graph.make_node( - 'Cast', - inputs=[node_img_width], - to=dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)]) - - node_img_height_cast = graph.make_node( - 'Cast', - inputs=[node_img_height], - to=dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)]) - - cls.node_pred_box_x1_decode = graph.make_node( - 'Mul', - inputs=[node_pred_box_x1, node_img_width_cast]) #boxes[box_idx] - - cls.node_pred_box_y1_decode = graph.make_node( - 'Mul', inputs=[node_pred_box_y1, - node_img_height_cast]) #boxes[box_idx + 1] - - cls.node_pred_box_x2_decode = graph.make_node( - 'Mul', - inputs=[node_pred_box_x2, node_img_width_cast]) #boxes[box_idx + 2] - - cls.node_pred_box_y2_decode = graph.make_node( - 'Mul', inputs=[node_pred_box_y2, - node_img_height_cast]) #boxes[box_idx + 3] - - if node.attr('clip_bbox'): - node_number_one = graph.make_node( - 'Constant', - inputs=[], - dtype=dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)], - value=[1]) - - node_new_img_height = graph.make_node( - 'Sub', inputs=[node_img_height_cast, node_number_one]) - - node_new_img_width = graph.make_node( - 'Sub', inputs=[node_img_width_cast, node_number_one]) - - cls.node_pred_box_x2_sub_w = graph.make_node( - 'Sub', - inputs=[cls.node_pred_box_x2_decode, node_new_img_width]) - - cls.node_pred_box_y2_sub_h = graph.make_node( - 'Sub', - inputs=[cls.node_pred_box_y2_decode, node_new_img_height]) - - node_pred_box_x1_clip = mapper_helper.clip_helper( - graph, node, cls.node_pred_box_x1_decode, - float(MAX_FLOAT32), 0.0) - node_pred_box_y1_clip = mapper_helper.clip_helper( - graph, node, cls.node_pred_box_y1_decode, - float(MAX_FLOAT32), 0.0) - node_pred_box_x2_clip = mapper_helper.clip_helper( - graph, node, cls.node_pred_box_x2_sub_w, - float(MAX_FLOAT32), 0.0) - node_pred_box_y2_clip = mapper_helper.clip_helper( - graph, node, cls.node_pred_box_y2_sub_h, - float(MAX_FLOAT32), 0.0) - node_pred_box_x2_res = graph.make_node( - 'Sub', - inputs=[cls.node_pred_box_x2_decode, node_pred_box_x2_clip]) - - node_pred_box_y2_res = graph.make_node( - 'Sub', - inputs=[cls.node_pred_box_y2_decode, node_pred_box_y2_clip]) - - node_pred_box_result = graph.make_node( - 'Concat', - inputs=[ - node_pred_box_x1_clip, node_pred_box_y1_clip, - node_pred_box_x2_res, node_pred_box_y2_res - ], - outputs=node.output('Boxes'), - axis=-1) - else: - node_pred_box_result = graph.make_node( - 'Concat', - inputs=[ - cls.node_pred_box_x1_decode, cls.node_pred_box_y1_decode, - cls.node_pred_box_x2_decode, cls.node_pred_box_y2_decode - ], - outputs=node.output('Boxes'), - axis=-1) - node_score_shape = graph.make_node( - "Constant", - inputs=[], - dtype=dtypes.ONNX.INT64, - value=cls.score_shape) - - node_score_new_shape = graph.make_node( - 'Reshape', - inputs=[cls.node_score, node_score_shape], - outputs=node.output('Scores')) diff --git a/paddle2onnx/legacy/op_mapper/logic.py b/paddle2onnx/legacy/op_mapper/logic.py deleted file mode 100755 index 6e2c198fa3a..00000000000 --- a/paddle2onnx/legacy/op_mapper/logic.py +++ /dev/null @@ -1,280 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import numpy as np -from paddle2onnx.legacy.constant import dtypes -from paddle2onnx.legacy.op_mapper import OpMapper as op_mapper -import paddle -from paddle2onnx.utils import logging -from paddle2onnx.legacy.op_mapper import mapper_helper - - -@op_mapper('greater_equal') -class GreaterOrEqual(): - support_opset_version_range = (12, 15) - - @classmethod - def opset_12(cls, graph, node, **kw): - onnx_node = graph.make_node( - 'GreaterOrEqual', - inputs=[node.input('X', 0), node.input('Y', 0)], - outputs=node.output('Out')) - - -@op_mapper('equal') -class Equal(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - if node.input_dtype('X', 0) in [paddle.float32, paddle.float64]: - warning_info = "Operator 'Equal' only support input with dtype of int/bool, now the dtype of input is {}, this may cause wrong results, it is more recommend converting this model with opset version >= 11.".format( - node.input_dtype('X', 0)) - logging.warning(warning_info) - x_node = graph.make_node( - 'Cast', inputs=node.input('X'), to=dtypes.ONNX.INT32) - y_node = graph.make_node( - 'Cast', inputs=node.input('Y'), to=dtypes.ONNX.INT32) - onnx_node = graph.make_node( - 'Equal', inputs=[x_node, y_node], outputs=node.output('Out')) - else: - onnx_node = graph.make_node( - 'Equal', - inputs=[node.input('X', 0), node.input('Y', 0)], - outputs=node.output('Out')) - - @classmethod - def opset_11(cls, graph, node, **kw): - onnx_node = graph.make_node( - 'Equal', - inputs=[node.input('X', 0), node.input('Y', 0)], - outputs=node.output('Out')) - - -@op_mapper('not_equal') -class NotEqual(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - equal_val = None - if node.input_dtype('X', 0) in [paddle.float32, paddle.float64]: - warning_info = "Operator 'not_equal' only support input with dtype of int/bool, now the dtype of input is {}, this may cause wrong results, it is more recommend converting this model with opset version >= 11.".format( - node.input_dtype('X', 0)) - logging.warning(warning_info) - x_node = graph.make_node( - 'Cast', inputs=node.input('X'), to=dtypes.ONNX.INT32) - y_node = graph.make_node( - 'Cast', inputs=node.input('Y'), to=dtypes.ONNX.INT32) - equal_val = graph.make_node( - 'Equal', inputs=[x_node, y_node], outputs=node.output('Out')) - else: - equal_val = graph.make_node( - 'Equal', - inputs=[node.input('X', 0), node.input('Y', 0)], - outputs=node.output('Out')) - k_node = graph.make_node( - 'Cast', inputs=[equal_val], to=dtypes.ONNX.INT64) - const = graph.make_node('Constant', dtype=dtypes.ONNX.INT64, value=1) - sub_ = graph.make_node('Sub', inputs=[const, k_node]) - graph.make_node( - 'Cast', - inputs=[sub_], - outputs=node.output('Out'), - to=dtypes.ONNX.BOOL) - - @classmethod - def opset_11(cls, graph, node, **kw): - equal_val = graph.make_node( - 'Equal', inputs=[node.input('X', 0), node.input('Y', 0)]) - k_node = graph.make_node( - 'Cast', inputs=[equal_val], to=dtypes.ONNX.INT64) - const = graph.make_node('Constant', dtype=dtypes.ONNX.INT64, value=1) - sub_ = graph.make_node('Sub', inputs=[const, k_node]) - graph.make_node( - 'Cast', - inputs=[sub_], - outputs=node.output('Out'), - to=dtypes.ONNX.BOOL) - - -@op_mapper('greater_than') -class GreaterThan(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - if node.input_dtype('X', 0) in [paddle.int32, paddle.int64]: - warning_info = "Operator 'greater_than' only support input with dtype of float/double, now the dtype of input is {}, this may cause wrong results, it is more recommend converting this model with opset version >= 11.".format( - node.input_dtype('X', 0)) - logging.warning(warning_info) - x_node = graph.make_node( - 'Cast', inputs=node.input('X'), to=dtypes.ONNX.INT32) - y_node = graph.make_node( - 'Cast', inputs=node.input('Y'), to=dtypes.ONNX.INT32) - graph.make_node( - 'Greater', - inputs=[node.input('X', 0), node.input('Y', 0)], - outputs=node.output('Out')) - else: - graph.make_node( - 'Greater', - inputs=[node.input('X', 0), node.input('Y', 0)], - outputs=node.output('Out')) - - @classmethod - def opset_11(cls, graph, node, **kw): - onnx_node = graph.make_node( - 'Greater', - inputs=[node.input('X', 0), node.input('Y', 0)], - outputs=node.output('Out')) - - -@op_mapper('logical_and') -class LogicalAnd(): - support_opset_version_range = (1, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - onnx_node = graph.make_node( - 'And', - inputs=[node.input('X', 0), node.input('Y', 0)], - outputs=node.output('Out')) - - -@op_mapper('logical_not') -class LogicalNot(): - support_opset_version_range = (1, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - graph.make_node( - 'Not', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('logical_or') -class LogicalOr(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - graph.make_node( - 'Or', - inputs=[node.input('X', 0), node.input('Y', 0)], - outputs=node.output('Out')) - - -@op_mapper('logical_xor') -class LogicalXOr(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - graph.make_node( - 'Xor', - inputs=[node.input('X', 0), node.input('Y', 0)], - outputs=node.output('Out')) - - -@op_mapper('less_equal') -class LessOrEqual(): - support_opset_version_range = (12, 15) - - @classmethod - def opset_12(cls, graph, node, **kw): - onnx_node = graph.make_node( - 'LessOrEqual', - inputs=[node.input('X', 0), node.input('Y', 0)], - outputs=node.output('Out')) - - -@op_mapper('less_than') -class Less_than(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - if node.input_dtype('X', 0) in [paddle.int32, paddle.int64]: - warning_info = "Operator 'less_than' only support input with dtype of float/double, now the dtype of input is {}, this may cause wrong results, it is more recommend converting this model with opset version >= 11.".format( - node.input_dtype('X', 0)) - logging.warning(warning_info) - x_node = graph.make_node( - 'Cast', inputs=node.input('X'), to=dtypes.ONNX.INT32) - y_node = graph.make_node( - 'Cast', inputs=node.input('Y'), to=dtypes.ONNX.INT32) - graph.make_node( - 'Less', - inputs=[node.input('X', 0), node.input('Y', 0)], - outputs=node.output('Out')) - else: - graph.make_node( - 'Less', - inputs=[node.input('X', 0), node.input('Y', 0)], - outputs=node.output('Out')) - - @classmethod - def opset_9(cls, graph, node, **kw): - graph.make_node( - 'Less', - inputs=[node.input('X', 0), node.input('Y', 0)], - outputs=node.output('Out'), ) - - -@op_mapper('isfinite_v2') -class Isfinite(): - support_opset_version_range = (10, 15) - - @classmethod - def opset_10(cls, graph, node, **kw): - is_inf = graph.make_node('IsInf', inputs=node.input('X', 0)) - is_nan = graph.make_node('IsNaN', inputs=node.input('X', 0)) - finite = graph.make_node('Or', inputs=[is_inf, is_nan]) - graph.make_node('Not', inputs=[finite], outputs=node.output('Out')) - - -@op_mapper('isinf_v2') -class IsInf(): - support_opset_version_range = (10, 15) - - @classmethod - def opset_10(cls, graph, node, **kw): - graph.make_node( - 'IsInf', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('isnan_v2') -class IsNaN(): - support_opset_version_range = (9, 15) - - @classmethod - def opset_9(cls, graph, node, **kw): - graph.make_node( - 'IsNaN', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('isnan') -class IsNaN(): - support_opset_version_range = (9, 15) - - @classmethod - def opset_9(cls, graph, node, **kw): - isnan = graph.make_node('IsNaN', inputs=node.input('X')) - cast_node = graph.make_node( - 'Cast', inputs=isnan, attrs={'to': dtypes.ONNX.FLOAT}) - reduce_node = graph.make_node( - 'ReduceMax', inputs=[cast_node], keepdims=False) - mapper_helper.unsqueeze_helper(graph, reduce_node, [0], - node.output('Out')) diff --git a/paddle2onnx/legacy/op_mapper/mapper_helper.py b/paddle2onnx/legacy/op_mapper/mapper_helper.py deleted file mode 100755 index 9850be9012c..00000000000 --- a/paddle2onnx/legacy/op_mapper/mapper_helper.py +++ /dev/null @@ -1,432 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -import paddle.fluid.core as core -import six -import copy -from paddle2onnx.legacy.constant import dtypes -import paddle -from onnx import TensorProto - - -def is_static_shape(shape): - if len(shape) > 1 and shape[1:].count(-1) > 0: - raise Exception( - "Converting this model to ONNX need with static input shape," \ - " please fix input shape of this model, see doc Q2 in" \ - " https://github.com/PaddlePaddle/paddle2onnx/blob/develop/docs/en/FAQ.md." - ) - - -def shape_helper(graph, input, dim=None): - if dim is None: - shape_node = graph.make_node('Shape', inputs=[input]) - return shape_node - full_shape = graph.make_node('Shape', inputs=[input]) - shape_node = slice_helper(graph, full_shape, [0], [dim], [dim + 1]) - return shape_node - - -def unsqueeze_helper(graph, input, axes, outputs=None): - inputs = [] - if not isinstance(input, list): - input = [input] - inputs.append(input[0]) - if not isinstance(axes, list): - axes = [axes] - if outputs is not None and isinstance(outputs, six.string_types): - outputs = [outputs] - - if graph.opset_version < 13: - unsqueeze_node = graph.make_node( - "Unsqueeze", inputs=inputs, outputs=outputs, axes=axes) - return unsqueeze_node - else: - axes_node = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=axes) - inputs = inputs + [axes_node] - unsqueeze_node = graph.make_node( - "Unsqueeze", inputs=inputs, outputs=outputs) - return unsqueeze_node - - -def split_helper(graph, input, axis=0, split=None, outputs=None): - assert outputs is not None, "outputs can not be None in split_helper." - inputs = [] - if not isinstance(input, list): - input = [input] - inputs.append(input[0]) - if split is not None and not isinstance(split, list): - split = [split] - if split is None: - split_node = graph.make_node( - "Split", inputs=inputs, outputs=outputs, axis=axis) - return split_node - if graph.opset_version < 13: - split_node = graph.make_node( - "Split", inputs=inputs, outputs=outputs, axis=axis, split=split) - return split_node - else: - split = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=split) - inputs = inputs + [split] - split_node = graph.make_node( - "Split", inputs=inputs, axis=axis, outputs=outputs) - return split_node - - -def slice_helper(graph, - input, - axes, - starts, - ends, - outputs=None, - dtype=dtypes.ONNX.INT64): - inputs = [] - if not isinstance(input, list): - input = [input] - inputs.append(input[0]) - if axes is not None and not isinstance(axes, list): - axes = [axes] - if starts is not None and not isinstance(starts, (list, six.string_types)): - starts = [starts] - if ends is not None and not isinstance(ends, (list, six.string_types)): - ends = [ends] - - if graph.opset_version < 10: - attrs = { - 'starts': starts, - 'ends': ends, - } - if axes not in [None, []]: - attrs['axes'] = axes - slice_node = graph.make_node( - "Slice", inputs=inputs, outputs=outputs, attrs=attrs) - return slice_node - else: - if not isinstance(starts, six.string_types): - starts = graph.make_node('Constant', dtype=dtype, value=starts) - if not isinstance(ends, six.string_types): - ends = graph.make_node('Constant', dtype=dtype, value=ends) - inputs = inputs + [starts, ends] - if axes not in [None, []]: - axes_node = graph.make_node('Constant', dtype=dtype, value=axes) - inputs.append(axes_node) - slice_node = graph.make_node("Slice", inputs=inputs, outputs=outputs) - return slice_node - - -def squeeze_helper(graph, input, axes=None, outputs=None): - inputs = [] - if not isinstance(input, list): - input = [input] - inputs.append(input[0]) - if axes is not None and not isinstance(axes, list): - axes = [axes] - if graph.opset_version < 13: - squeeze_node = graph.make_node( - "Squeeze", inputs=inputs, axes=axes, outputs=outputs) - return squeeze_node - else: - if axes is not None: - axes_node = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=axes) - inputs.append(axes_node) - squeeze_node = graph.make_node( - "Squeeze", inputs=inputs, outputs=outputs) - return squeeze_node - - -def unsqueeze_helper(graph, input, axes, outputs=None): - inputs = [] - if isinstance(input, list): - input = input[0] - inputs.append(input) - if not isinstance(axes, list): - axes = [axes] - if graph.opset_version < 13: - unsqueeze_node = graph.make_node( - 'Unsqueeze', inputs=inputs, axes=axes, outputs=outputs) - else: - axes_node = graph.make_node( - 'Constant', attrs={'dtype': dtypes.ONNX.INT64, - 'value': axes}) - inputs.append(axes_node) - unsqueeze_node = graph.make_node( - 'Unsqueeze', inputs=inputs, outputs=outputs) - return unsqueeze_node - - -def split_helper(graph, inputs, outputs, axis, split, dtype=paddle.float32): - if not isinstance(inputs, (list, tuple)): - inputs = [inputs] - - if not isinstance(outputs, int) and not isinstance(outputs, (list, tuple)): - outputs = [outputs] - - if dtype == paddle.float64: - cast_inputs = [] - for i in range(len(inputs)): - one = graph.make_node( - 'Cast', inputs=[inputs[i]], to=TensorProto.FLOAT) - cast_inputs.append(one) - if graph.opset_version < 13: - split_node = graph.make_node( - "Split", - inputs=cast_inputs, - outputs=outputs, - axis=axis, - split=split) - else: - split_const = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=split) - split_node = graph.make_node( - "Split", - inputs=cast_inputs + [split_const], - outputs=outputs, - axis=axis) - casted_output = [] - for i in range(len(outputs)): - one = graph.make_node( - 'Cast', - inputs=[split_node[i]], - outputs=[outputs[i]], - to=TensorProto.DOUBLE) - casted_output.append(one) - return casted_output - else: - if graph.opset_version < 13: - split_node = graph.make_node( - "Split", inputs=inputs, outputs=outputs, axis=axis, split=split) - else: - split_const = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=split) - split_node = graph.make_node( - "Split", - inputs=inputs + [split_const], - outputs=outputs, - axis=axis) - return split_node - - -def constant_helper(graph, dtype, value, shape=None, outputs=[]): - constant = graph.make_node( - 'Constant', - inputs=[], - outputs=outputs, - attrs={ - 'dims': shape, - 'dtype': dtypes.DTYPE_PADDLE_ONNX_MAP[dtype], - 'value': value - }) - return constant - - -def clip_helper(graph, node, input, max, min, output=[]): - x_dtype = node.input_dtype('X', 0) - if (isinstance(min, six.string_types) or - isinstance(max, six.string_types)) and graph.opset_version < 11: - raise Exception( - "min or max of Clip is Tensor, please try with higher onnx opset_version." - ) - if graph.opset_version < 11: - if x_dtype != paddle.float32: - input = graph.make_node( - 'Cast', inputs=[input], to=dtypes.ONNX.FLOAT) - clip = graph.make_node('Clip', inputs=input, max=max, min=min) - clip = graph.make_node( - 'Cast', - inputs=[clip], - to=dtypes.DTYPE_PADDLE_ONNX_MAP[x_dtype], - outputs=output) - else: - clip = graph.make_node( - 'Clip', inputs=input, max=max, min=min, outputs=output) - else: - if x_dtype != paddle.float32: - input = graph.make_node( - 'Cast', inputs=[input], to=dtypes.ONNX.FLOAT) - - if not isinstance(min, six.string_types): - min = graph.make_node( - 'Constant', - attrs={ - 'dtype': dtypes.DTYPE_PADDLE_ONNX_MAP[paddle.float32], - 'value': min - }) - else: - if node.input_dtype('Min', 0) != paddle.float32: - min = graph.make_node( - 'Cast', - inputs=min, - attrs={'to': dtypes.DTYPE_PADDLE_ONNX_MAP[paddle.float32]}) - min = graph.make_node('Squeeze', min) - - if not isinstance(max, six.string_types): - max = graph.make_node( - 'Constant', - attrs={ - 'dtype': dtypes.DTYPE_PADDLE_ONNX_MAP[paddle.float32], - 'value': max - }) - else: - if node.input_dtype('Max', 0) != paddle.float32: - max = graph.make_node( - 'Cast', - inputs=max, - attrs={'to': dtypes.DTYPE_PADDLE_ONNX_MAP[paddle.float32]}) - max = graph.make_node('Squeeze', max) - if x_dtype != paddle.float32: - clip_pre = graph.make_node('Clip', inputs=[input, min, max]) - clip = graph.make_node( - 'Cast', - inputs=[clip_pre], - outputs=output, - to=dtypes.DTYPE_PADDLE_ONNX_MAP[x_dtype]) - else: - clip = graph.make_node( - 'Clip', inputs=[input, min, max], outputs=output) - return clip - - -def dtype_alignment(graph, nodes, node_dtypes, to=None): - assert len(nodes) == len( - node_dtypes), "Length of nodes and node_dtypes should be equal." - dtype_order = [ - core.VarDesc.VarType.BOOL, - core.VarDesc.VarType.INT16, - core.VarDesc.VarType.INT32, - core.VarDesc.VarType.INT64, - core.VarDesc.VarType.FP16, - core.VarDesc.VarType.FP32, - core.VarDesc.VarType.FP64, - ] - max_index = -1 - for dtype in node_dtypes: - index = dtype_order.index(dtype) - if index > max_index: - max_index = index - - if max_index < 0: - return nodes - - casted_nodes = list() - cast_dtype = dtype_order[max_index] - cast_dtype = dtypes.DTYPE_PADDLE_ONNX_MAP[cast_dtype] - for i, dtype in enumerate(node_dtypes): - index = dtype_order.index(dtype) - if to is not None: - cast_dtype = to - condition = dtypes.DTYPE_PADDLE_ONNX_MAP[index] != cast_dtype - else: - condition = index != max_index - if condition: - cast_node = graph.make_node( - 'Cast', inputs=[nodes[i]], to=cast_dtype) - casted_nodes.append(cast_node) - else: - casted_nodes.append(nodes[i]) - return casted_nodes - - -def cast(graph, input, origin_dtype, target_dtype): - if not isinstance(origin_dtype, six.string_types): - origin_dtype = dtypes.DTYPE_PADDLE_STR_MAP[origin_dtype] - if origin_dtype != target_dtype: - cast_node = graph.make_node( - 'Cast', inputs=input, to=dtypes.DTYPE_ONNX_STR_MAP[target_dtype]) - return cast_node - return input - - -def shape_alignment(graph, nodes, node_shapes): - assert len(nodes) == len( - node_shapes), "Length of nodes and node_shapes should be equal." - max_dim = -1 - for shape in node_shapes: - dim = len(shape) - if dim > max_dim: - max_dim = dim - - if max_dim < 0: - return nodes - - assert max_dim == 1 or max_dim == 0, "max_dim is only supported when max_dim is 1 or 0." - max_dim = 1 if max_dim == 0 else max_dim - unsqueeze_nodes = list() - for i, shape in enumerate(node_shapes): - dim = len(shape) - if dim != max_dim: - unsqueeze_node = nodes[i] - for j in range(max_dim - dim): - unsqueeze_node = unsqueeze_helper(graph, unsqueeze_node, [0]) - unsqueeze_nodes.append(unsqueeze_node) - else: - unsqueeze_nodes.append(nodes[i]) - return unsqueeze_nodes - - -def get_tensor_list_node(graph, node, name, dtype=None): - node_list = node.input(name) - node_dtypes = [node.input_dtype(name, i) for i in range(len(node_list))] - node_list = dtype_alignment(graph, node_list, node_dtypes, dtype) - - node_shapes = [node.input_shape(name, i) for i in range(len(node_list))] - node_list = shape_alignment(graph, node_list, node_shapes) - node = graph.make_node("Concat", inputs=node_list, axis=0) - return node - - -def get_value_from_parameters(graph, input_node): - assert input_node in graph.parameters, "{} is not in graph.parameters".format( - input_node) - data = graph.parameters[input_node].attribute[0].t.int32_data - if data is None or len(data) < 1: - data = graph.parameters[input_node].attribute[0].t.int64_data - value = [val for _, val in enumerate(data)] - return value - - -# return value -# arg1: attr_value -# arg2: attr_value is tensor or not -def get_node_attr_value(graph, - node, - attr_name=None, - attr_tensor_name=None, - attr_tensor_list_name=None, - return_list=False, - dtype=None): - attr_tensor = node.input(attr_tensor_name) - attr_tensor_list = node.input(attr_tensor_list_name) - if attr_tensor is not None and len(attr_tensor) > 0: - value = node.input(attr_tensor_name)[0] - if return_list: - try: - value = get_value_from_parameters(graph, value) - return value, False # value, is_tensor - except Exception: - return value, True - else: - input_dtype = dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype( - attr_tensor_name, 0)] - if input_dtype != dtype: - value = graph.make_node('Cast', inputs=[value], to=dtype) - return value, True - elif attr_tensor_list is not None and len(attr_tensor_list) > 0: - value = get_tensor_list_node(graph, node, attr_tensor_list_name, dtype) - return value, True - else: - value = node.attr(attr_name) - return value, False diff --git a/paddle2onnx/legacy/op_mapper/math.py b/paddle2onnx/legacy/op_mapper/math.py deleted file mode 100755 index ff94f93ff9d..00000000000 --- a/paddle2onnx/legacy/op_mapper/math.py +++ /dev/null @@ -1,1273 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import numpy as np -from paddle2onnx.legacy.constant import dtypes -from paddle2onnx.legacy.op_mapper import OpMapper as op_mapper -from paddle2onnx.legacy.op_mapper import mapper_helper -import paddle - - -@op_mapper('matmul') -class MatMul(): - support_opset_version_range = (1, 12) - - @classmethod - def opset_1(cls, graph, node, **kw): - x = node.input('X', idx=0) - y = node.input('Y', idx=0) - if node.attr('transpose_X'): - perm = list(range(len(node.input_shape('X', 0)))) - perm[-1], perm[-2] = perm[-2], perm[-1] - if node.input_dtype('X', 0) == paddle.float64: - x = graph.make_node('Cast', inputs=x, to=dtypes.ONNX.FLOAT) - x = graph.make_node('Transpose', inputs=[x], perm=perm) - if node.attr('transpose_Y'): - perm = list(range(len(node.input_shape('Y', 0)))) - perm[-1], perm[-2] = perm[-2], perm[-1] - if node.input_dtype('Y', 0) == paddle.float64: - y = graph.make_node('Cast', inputs=y, to=dtypes.ONNX.FLOAT) - y = graph.make_node('Transpose', inputs=[y], perm=perm) - if node.attr('alpha') == 1.0: - if node.input_dtype('X', 0) == paddle.float64: - output_node = graph.make_node('MatMul', inputs=[x, y]) - graph.make_node( - 'Cast', - inputs=output_node, - to=dtypes.ONNX.DOUBLE, - outputs=node.output('Out')) - else: - graph.make_node( - 'MatMul', inputs=[x, y], outputs=node.output('Out')) - else: - if node.input_dtype('X', 0) == paddle.float64: - output_node = graph.make_node('MatMul', inputs=[x, y]) - matmul = graph.make_node( - 'Cast', inputs=output_node, to=dtypes.ONNX.DOUBLE) - scale = graph.make_node( - 'Constant', - dtype=dtypes.ONNX.DOUBLE, - value=node.attr('alpha')) - else: - matmul = graph.make_node('MatMul', inputs=[x, y]) - scale = graph.make_node( - 'Constant', - dtype=dtypes.ONNX.FLOAT, - value=node.attr('alpha')) - - onnx_node = graph.make_node( - 'Mul', inputs=[matmul, scale], outputs=node.output('Out')) - - -@op_mapper('matmul_v2') -class MatMul(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - x = node.input('X', idx=0) - y = node.input('Y', idx=0) - out = node.output('Out') - ## TODO(wangjunjie06): The current addition of cast op is only for onnxruntime optimization, after onnxruntime is repaired, remove this logic - if node.attr('trans_x'): - perm = list(range(len(node.input_shape('X', 0)))) - perm[-1], perm[-2] = perm[-2], perm[-1] - if node.input_dtype('X', 0) == paddle.float64: - x = graph.make_node('Cast', inputs=x, to=dtypes.ONNX.FLOAT) - x = graph.make_node('Transpose', inputs=[x], perm=perm) - if node.attr('trans_y'): - perm = list(range(len(node.input_shape('Y', 0)))) - perm[-1], perm[-2] = perm[-2], perm[-1] - if node.input_dtype('Y', 0) == paddle.float64: - y = graph.make_node('Cast', inputs=y, to=dtypes.ONNX.FLOAT) - y = graph.make_node('Transpose', inputs=[y], perm=perm) - if node.input_dtype('X', 0) == paddle.float64: - output_node = graph.make_node('MatMul', inputs=[x, y]) - graph.make_node( - 'Cast', inputs=output_node, to=dtypes.ONNX.DOUBLE, outputs=out) - else: - graph.make_node('MatMul', inputs=[x, y], outputs=out) - - -@op_mapper('exp') -class Exp(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - graph.make_node( - 'Exp', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('abs') -class Abs: - support_opset_version_range = (7, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - graph.make_node( - 'Abs', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('erf') -class Erf(): - support_opset_version_range = (9, 15) - - @classmethod - def opset_9(cls, graph, node, **kw): - x_dtype = node.input_dtype('X', 0) - x = node.input('X', 0) - # onnxruntime only support float32 Erf - if x_dtype != paddle.float32: - x = graph.make_node('Cast', inputs=x, to=dtypes.ONNX.FLOAT) - erf_node = graph.make_node('Erf', inputs=[x]) - graph.make_node( - 'Cast', - inputs=[erf_node], - to=dtypes.DTYPE_PADDLE_ONNX_MAP[x_dtype], - outputs=node.output('Out')) - else: - graph.make_node('Erf', inputs=[x], outputs=node.output('Out')) - - -@op_mapper('acos') -class Acos(): - supports_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - graph.make_node( - 'Acos', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('asin') -class Asin(): - supports_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - graph.make_node( - 'Asin', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('sinh') -class Sinh(): - supports_opset_version_range = (9, 15) - - @classmethod - def opset_9(cls, graph, node, **kw): - graph.make_node( - 'Sinh', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('sin') -class Sin(): - supports_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - graph.make_node( - 'Sin', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('atan') -class Atan(): - supports_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - graph.make_node( - 'Atan', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('tan') -class Tan(): - supports_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - graph.make_node( - 'Tan', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('ceil') -class Ceil(): - supports_opset_version_range = (7, 15) - - @classmethod - def opset_6(cls, graph, node, **kw): - graph.make_node( - 'Ceil', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('cos') -class Cos(): - supports_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - graph.make_node( - 'Cos', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('cosh') -class Cosh(): - supports_opset_version_range = (9, 15) - - @classmethod - def opset_9(cls, graph, node, **kw): - graph.make_node( - 'Cosh', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('log2') -class Log2(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - _ln2 = 0.693147180559945309 - dtype = dtypes.ONNX.FLOAT - if node.input_dtype('X', 0) == paddle.float64: - dtype = dtypes.ONNX.DOUBLE - _ln2 = graph.make_node('Constant', dtype=dtype, value=_ln2) - lnx = graph.make_node('Log', inputs=node.input('X')) - graph.make_node('Div', inputs=[lnx, _ln2], outputs=node.output('Out')) - - -@op_mapper('logsumexp') -class LogSumExp(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - - if node.attr('reduce_all'): - if not node.attr('keepdim'): - reduce_node = graph.make_node( - 'ReduceLogSumExp', - inputs=node.input('X'), - keepdims=node.attr('keepdim')) - mapper_helper.unsqueeze_helper(graph, reduce_node, [0], - node.output('Out')) - else: - graph.make_node( - 'ReduceLogSumExp', - inputs=node.input('X'), - keepdims=node.attr('keepdim'), - outputs=node.output('Out')) - else: - graph.make_node( - 'ReduceLogSumExp', - inputs=node.input('X'), - keepdims=node.attr('keepdim'), - axes=node.attr('axis'), - outputs=node.output('Out')) - - -@op_mapper( - [ - 'elementwise_add', 'elementwise_sub', 'elementwise_div', - 'elementwise_mul', 'elementwise_min', 'elementwise_max', - 'elementwise_pow' - ], - mapper_dict={ - 'elementwise_add': 'Add', - 'elementwise_sub': 'Sub', - 'elementwise_div': 'Div', - 'elementwise_mul': 'Mul', - 'elementwise_min': 'Min', - 'elementwise_max': 'Max', - 'elementwise_pow': 'Pow' - }) -class ElementwiseOps(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - x_shape = node.input_shape('X', 0) - y_shape = node.input_shape('Y', 0) - if node.type in ["elementwise_min", "elementwise_max"]: - assert False, "{} op is not supported when opset version < 8".format( - node.type) - - op_type = kw['mapper_dict'][node.type] - axis = node.attr('axis') - x = node.input('X', 0) - y = node.input('Y', 0) - - if axis == -1 or axis == (len(x_shape) - 1 - ) or len(x_shape) == len(y_shape): - onnx_node = graph.make_node( - op_type, inputs=[x, y], outputs=node.output('Out')) - else: - broadcast_shape = [1] * len(x_shape) - broadcast_shape[axis:axis + len(y_shape)] = y_shape - broadcast_shape_node = graph.make_node( - 'Constant', - dtype=dtypes.ONNX.INT64, - value=list(broadcast_shape)) - y_node = graph.make_node( - 'Reshape', inputs=[y, broadcast_shape_node]) - onnx_node = graph.make_node( - op_type, inputs=[x, y_node], outputs=node.output('Out')) - - @classmethod - def opset_8(cls, graph, node, **kw): - op_type = kw['mapper_dict'][node.type] - x = node.input('X', 0) - y = node.input('Y', 0) - axis = node.attr('axis') - x_shape = node.input_shape('X', 0) - y_shape = node.input_shape('Y', 0) - if axis == -1 or axis == (len(x_shape) - 1 - ) or len(x_shape) == len(y_shape): - onnx_node = graph.make_node( - op_type, inputs=[x, y], outputs=node.output('Out')) - else: - broadcast_shape = [1] * len(x_shape) - broadcast_shape[axis:axis + len(y_shape)] = y_shape - broadcast_shape_node = graph.make_node( - 'Constant', - dtype=dtypes.ONNX.INT64, - value=list(broadcast_shape)) - y_node = graph.make_node( - 'Reshape', inputs=[y, broadcast_shape_node]) - onnx_node = graph.make_node( - op_type, inputs=[x, y_node], outputs=node.output('Out')) - - -@op_mapper('elementwise_mod') -class ElementWiseMod(): - support_opset_version_range = (10, 15) - - @classmethod - def opset_10(cls, graph, node, **kw): - x_shape = node.input_shape('X', 0) - y_shape = node.input_shape('Y', 0) - axis = node.attr('axis') - x = node.input('X', 0) - y = node.input('Y', 0) - - if node.input_dtype('Y', 0) == paddle.int32 or node.input_dtype( - 'Y', 0) == paddle.int64: - onnx_node = graph.make_node( - "Mod", inputs=[x, y], outputs=node.output('Out')) - return - - fmod = 1 - - abs_x_node = graph.make_node("Abs", inputs=[x]) - abs_y_node = graph.make_node("Abs", inputs=[y]) - - dtype = dtypes.ONNX.FLOAT - val_0 = [0.0] - val_1 = [-1.0] - if node.input_dtype('Y', 0) == paddle.float64: - dtype = dtypes.ONNX.DOUBLE - if node.input_dtype('Y', 0) == paddle.int32: - dtype = dtypes.ONNX.INT32 - val_0 = [0] - val_1 = [-1] - if node.input_dtype('Y', 0) == paddle.int64: - dtype = dtypes.ONNX.INT64 - val_0 = [0] - val_1 = [-1] - zero_node = graph.make_node('Constant', dtype=dtype, value=val_0) - one_node = graph.make_node('Constant', dtype=dtype, value=val_1) - - mod_node = graph.make_node( - "Mod", inputs=[abs_x_node, abs_y_node], fmod=fmod) - - minus_node = graph.make_node("Mul", inputs=[mod_node, one_node]) - - condition_dtype = graph.make_node("Less", inputs=[x, zero_node]) - condition = graph.make_node( - 'Cast', inputs=[condition_dtype], to=dtypes.ONNX.BOOL) - - mod_res = graph.make_node( - "Where", inputs=[condition, minus_node, mod_node]) - - add_node = graph.make_node("Add", inputs=[mod_res, y]) - - mod_y_mul_node = graph.make_node("Mul", inputs=[mod_res, y]) - condition_dtype_1 = graph.make_node( - "Less", inputs=[mod_y_mul_node, zero_node]) - condition_1 = graph.make_node( - 'Cast', inputs=[condition_dtype_1], to=dtypes.ONNX.BOOL) - - graph.make_node( - "Where", - inputs=[condition_1, add_node, mod_res], - outputs=node.output('Out')) - - -@op_mapper('elementwise_floordiv') -class ElementWiseFloorDiv(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - x = node.input('X', 0) - y = node.input('Y', 0) - axis = node.attr('axis') - x_shape = node.input_shape('X', 0) - y_shape = node.input_shape('Y', 0) - x_dtype = node.input_dtype('X', 0) - y_dtype = node.input_dtype('Y', 0) - x_dtype = dtypes.DTYPE_PADDLE_STR_MAP[x_dtype] - y_dtype = dtypes.DTYPE_PADDLE_STR_MAP[y_dtype] - is_int = False - if x_dtype.count('int') > 0 and y_dtype.count('int') > 0: - is_int = True - if axis == -1 or axis == (len(x_shape) - 1 - ) or len(x_shape) == len(y_shape): - if is_int: - graph.make_node( - 'Div', inputs=[x, y], outputs=node.output('Out')) - else: - div_node = graph.make_node('Div', inputs=[x, y]) - graph.make_node( - 'Floor', inputs=[div_node], outputs=node.output('Out')) - else: - broadcast_shape = [1] * len(x_shape) - broadcast_shape[axis:axis + len(y_shape)] = y_shape - broadcast_shape_node = graph.make_node( - 'Constant', - dtype=dtypes.ONNX.INT64, - value=list(broadcast_shape)) - y_node = graph.make_node( - 'Reshape', inputs=[y, broadcast_shape_node]) - if is_int: - div_node = graph.make_node( - 'Div', inputs=[x, y_node], outputs=node.output('Out')) - else: - div_node = graph.make_node('Div', inputs=[x, y_node]) - graph.make_node( - 'Floor', inputs=[div_node], outputs=node.output('Out')) - - -@op_mapper('pow') -class Pow(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - x = node.input('X', 0) - x_dtype = node.input_dtype('X', 0) - factor = node.attr('factor') - # Pow-7 Only support input type as float and double - if x_dtype == paddle.int32 or x_dtype == paddle.int64: - x = graph.make_node('Cast', inputs=[x], to=dtypes.ONNX.FLOAT) - factor_node = graph.make_node( - 'Constant', - inputs=[], - dims=[1], - dtype=dtypes.ONNX.FLOAT, - value=factor) - else: - factor_node = graph.make_node( - 'Constant', - inputs=[], - dims=[1], - dtype=dtypes.DTYPE_PADDLE_ONNX_MAP[x_dtype], - value=factor) - if x_dtype == paddle.int32 or x_dtype == paddle.int64: - pow_node = graph.make_node('Pow', inputs=[x, factor_node]) - graph.make_node( - 'Cast', - inputs=[pow_node], - to=dtypes.DTYPE_PADDLE_ONNX_MAP[x_dtype], - outputs=node.output('Out')) - else: - graph.make_node( - 'Pow', inputs=[x, factor_node], outputs=node.output('Out')) - - @classmethod - def opset_12(cls, graph, node, **kw): - x = node.input('X', 0) - factor = node.attr('factor') - factor_node = graph.make_node( - 'Constant', - inputs=[], - dims=[1], - dtype=dtypes.ONNX.FLOAT, - value=factor) - pow_node = graph.make_node( - 'Pow', inputs=[x, factor_node], outputs=node.output('Out')) - - -@op_mapper('square') -class Square(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - x = node.input('X', 0) - onnx_node = graph.make_node( - 'Mul', inputs=[x, x], outputs=node.output('Out')) - - -@op_mapper('cumsum') -class CumSum(): - support_opset_version_range = (11, 15) - - @classmethod - def opset_11(cls, graph, node, **kw): - - axis = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=node.attr('axis')) - graph.make_node( - 'CumSum', - inputs=[node.input('X', 0), axis], - outputs=node.output('Out')) - - -@op_mapper('mul') -class Mul(): - support_opset_version_range = (5, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - x = node.input('X', 0) - y = node.input('Y', 0) - out = node.output('Out', 0) - x_num_col_dims = node.attr('x_num_col_dims') - y_num_col_dims = node.attr('y_num_col_dims') - flatten_x = graph.make_node( - 'Flatten', inputs=node.input('X'), attrs={'axis': x_num_col_dims}) - flatten_y = graph.make_node( - 'Flatten', inputs=node.input('Y'), attrs={'axis': y_num_col_dims}) - mul_node = graph.make_node('MatMul', inputs=[flatten_x, flatten_y]) - - x_shape = graph.make_node('Shape', inputs=[x]) - l_shape = mapper_helper.slice_helper( - graph, x_shape, axes=[0], starts=[0], ends=[x_num_col_dims]) - y_shape = graph.make_node('Shape', inputs=[y]) - y_rank = len(node.input_shape('Y', 0)) - r_shape = mapper_helper.slice_helper( - graph, y_shape, axes=[0], starts=[y_num_col_dims], ends=[y_rank]) - - out_shape = graph.make_node('Concat', inputs=[l_shape, r_shape], axis=0) - graph.make_node('Reshape', [mul_node, out_shape], node.output('Out')) - - -@op_mapper('affine_channel') -class AffineChannel(): - support_opset_version_range = (1, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - if "data_layout" in node.attrs.keys(): - assert node.attrs['data_layout'] == 'NCHW' or node.attrs['data_layout'] == "AnyLayout", \ - "The affine_channel data format should be 'NCHW', but received data format " \ - "is %s." % node.attrs['data_layout'] - x = node.input('X', 0) - bias = node.input('Bias', 0) - scale = node.input('Scale', 0) - scale = mapper_helper.unsqueeze_helper(graph, scale, [0, 2, 3]) - bias = mapper_helper.unsqueeze_helper(graph, bias, [0, 2, 3]) - x = graph.make_node('Mul', inputs=[x, scale]) - x = graph.make_node('Add', inputs=[x, bias], outputs=node.output('Out')) - - -@op_mapper('bmm') -class BMM(): - support_opset_version_range = (1, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - x = node.input('X', 0) - y = node.input('Y', 0) - mul_node = graph.make_node( - 'MatMul', inputs=[x, y], outputs=node.output('Out')) - - -@op_mapper('p_norm') -class PNorm(): - support_opset_version_range = (1, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - x = node.input('X', 0) - axis = node.attr('axis') - if isinstance(axis, (int, float)): - axis = [axis] - p = node.attr('porder') - keepdim = node.attr('keepdim') - dtype = dtypes.ONNX.FLOAT - if node.input_dtype('X', 0) == paddle.float64: - dtype = dtypes.ONNX.DOUBLE - - pnode = graph.make_node('Constant', dtype=dtype, value=[p]) - - abs_node = graph.make_node('Abs', inputs=[x]) - pow_node = graph.make_node('Pow', inputs=[abs_node, pnode]) - reduce_sum = graph.make_node( - 'ReduceSum', inputs=[pow_node], axes=axis, keepdims=keepdim) - pnode1 = graph.make_node('Constant', dtype=dtype, value=[1.0 / p]) - graph.make_node( - 'Pow', inputs=[reduce_sum, pnode1], outputs=node.output('Out')) - - @classmethod - def opset_13(cls, graph, node, **kw): - x = node.input('X', 0) - axis = node.attr('axis') - if isinstance(axis, (int, float)): - axis = [axis] - p = node.attr('porder') - keepdim = node.attr('keepdim') - pnode = graph.make_node('Constant', dtype=dtypes.ONNX.FLOAT, value=[p]) - abs_node = graph.make_node('Abs', inputs=[x]) - pow_node = graph.make_node('Pow', inputs=[abs_node, pnode]) - axes = graph.make_node('Constant', dtype=dtypes.ONNX.INT64, value=axis) - reduce_sum = graph.make_node( - 'ReduceSum', inputs=[pow_node, axes], keepdims=keepdim) - pnode1 = graph.make_node( - 'Constant', dtype=dtypes.ONNX.FLOAT, value=[1.0 / p]) - graph.make_node( - 'Pow', inputs=[reduce_sum, pnode1], outputs=node.output('Out')) - - -@op_mapper('sum') -class Sum(): - support_opset_version_range = (1, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - graph.make_node( - 'Sum', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('floor') -class Floor(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - graph.make_node( - 'Floor', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('log10') -class Log10(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - _ln10 = 2.30258509299404568401 - dtype = dtypes.ONNX.FLOAT - if node.input_dtype('X', 0) == paddle.float64: - dtype = dtypes.ONNX.DOUBLE - _ln10 = graph.make_node('Constant', dtype=dtype, value=_ln10) - lnx = graph.make_node('Log', inputs=node.input('X')) - graph.make_node('Div', inputs=[lnx, _ln10], outputs=node.output('Out')) - - -@op_mapper('log1p') -class Log1p(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - dtype = dtypes.ONNX.FLOAT - if node.input_dtype('X', 0) == paddle.float64: - dtype = dtypes.ONNX.DOUBLE - one = graph.make_node('Constant', attrs={'dtype': dtype, 'value': [1]}) - add_node = graph.make_node('Add', inputs=[node.input('X', 0), one]) - graph.make_node('Log', inputs=add_node, outputs=node.output('Out')) - - -@op_mapper( - ['reduce_all', 'reduce_any'], - mapper_dict={'reduce_all': 'ReduceMin', - 'reduce_any': 'ReduceMax'}) -class ReduceAll(): - support_opset_version_range = (6, 15) - - @classmethod - def opset_6(cls, graph, node, **kw): - op_type = kw['mapper_dict'][node.type] - input_dtype = node.block.vars[node.input('X', 0)].dtype - input_dtype = dtypes.DTYPE_PADDLE_ONNX_MAP[input_dtype] - all_node = graph.make_node( - 'Cast', inputs=[node.input('X', 0)], to=dtypes.ONNX.INT32) - - attrs = {'keepdims': node.attr('keep_dim'), } - if not node.attr('reduce_all'): - attrs['axes'] = node.attr('dim') - output_node = graph.make_node(op_type, inputs=[all_node], attrs=attrs) - - if node.attr('reduce_all') and not node.attr('keep_dim'): - output_node = mapper_helper.unsqueeze_helper(graph, output_node, - [0]) - graph.make_node( - 'Cast', - inputs=[output_node], - to=input_dtype, - outputs=node.output('Out')) - - -@op_mapper( - ['reduce_mean', 'reduce_sum', 'reduce_min', 'reduce_max', 'reduce_prod'], - mapper_dict={ - 'reduce_mean': 'ReduceMean', - 'reduce_sum': 'ReduceSum', - 'reduce_min': 'ReduceMin', - 'reduce_max': 'ReduceMax', - 'reduce_prod': 'ReduceProd' - }) -class ReduceMean(): - support_opset_version_range = (1, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - op_type = kw['mapper_dict'][node.type] - - output_shape = node.output_shape('Out', 0) - reduce_all = node.attr("reduce_all") - axes = node.attr("dim") - if reduce_all: - axes = list(range(len(node.input_shape("X", 0)))) - if len(axes) == len(node.input_shape("X", 0)): - reduce_all = True - keepdims = node.attr('keep_dim') - if keepdims: - cls.create_reduce_node(graph, op_type, - node.input("X"), node.output("Out"), axes, 1) - else: - if reduce_all: - shape = graph.make_node( - "Constant", dtype=dtypes.ONNX.INT64, value=[-1]) - flatten_node = graph.make_node( - "Reshape", inputs=[node.input("X")[0], shape]) - cls.create_reduce_node(graph, op_type, [flatten_node], - node.output("Out"), [0], 1) - else: - cls.create_reduce_node(graph, op_type, - node.input("X"), - node.output("Out"), axes, 0) - - @classmethod - def create_reduce_node(cls, graph, op_type, inputs, outputs, axes, - keepdims): - if graph.opset_version >= 13 and op_type == "ReduceSum": - axes = graph.make_node( - "Constant", dtype=dtypes.ONNX.INT64, value=axes) - output = graph.make_node( - "ReduceSum", - inputs=inputs + [axes], - outputs=outputs, - keepdims=keepdims) - else: - output = graph.make_node( - op_type, - inputs=inputs, - outputs=outputs, - axes=axes, - keepdims=keepdims) - - -@op_mapper('mean') -class Mean(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - shape = graph.make_node("Constant", dtype=dtypes.ONNX.INT64, value=[-1]) - flatten_node = graph.make_node( - "Reshape", inputs=[node.input("X")[0], shape]) - mean_node = graph.make_node( - 'ReduceMean', - inputs=flatten_node, - outputs=node.output("Out"), - keepdims=1) - - -@op_mapper('arg_max') -class ArgMax(): - support_opset_version_range = (1, 12) - - @classmethod - def opset_1(cls, graph, node, **kw): - if node.attr('dtype') and node.attr('dtype') == 2: - arg_node = graph.make_node( - 'ArgMax', - inputs=node.input('X'), - attrs={ - 'axis': node.attr('axis'), - 'keepdims': node.attr('keepdims') - }) - graph.make_node( - 'Cast', - inputs=arg_node, - attrs={'to': dtypes.ONNX.INT32}, - outputs=node.output('Out')) - else: - graph.make_node( - 'ArgMax', - inputs=node.input('X'), - outputs=node.output('Out'), - attrs={ - 'axis': node.attr('axis'), - 'keepdims': node.attr('keepdims') - }) - - -@op_mapper('arg_min') -class ArgMin(): - support_opset_version_range = (1, 12) - - @classmethod - def opset_1(cls, graph, node, **kw): - if node.attr('flatten'): - flatten = graph.make_node('Flatten', inputs=node.input('X'), axis=0) - squeeze_node = graph.make_node('Squeeze', inputs=flatten) - graph.make_node( - 'ArgMin', inputs=squeeze_node, outputs=node.output('Out')) - else: - if node.attr('keepdims'): - graph.make_node( - 'ArgMin', - inputs=node.input('X'), - outputs=node.output('Out'), - axis=node.attr('axis'), - keepdims=1) - else: - graph.make_node( - 'ArgMin', - inputs=node.input('X'), - outputs=node.output('Out'), - axis=node.attr('axis'), - keepdims=0) - - -@op_mapper('brelu') -class Hardtanh(): - support_opset_version_range = (9, 15) - - @classmethod - def opset_6(cls, graph, node, **kw): - mapper_helper.clip_helper(graph, node, - node.input('X', 0), - node.attr('t_max'), - node.attr('t_min'), node.output('Out', 0)) - - -@op_mapper('mv') -class Mv(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - graph.make_node( - 'MatMul', - inputs=[node.input('X', 0), node.input('Vec', 0)], - outputs=node.output('Out')) - - -@op_mapper('dot') -class Dot(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - mul_node = graph.make_node( - 'Mul', inputs=[node.input('X', 0), node.input('Y', 0)]) - graph.make_node( - 'ReduceSum', - inputs=[mul_node], - axes=[len(node.input_shape('X', 0)) - 1], - outputs=node.output('Out')) - - @classmethod - def opset_13(cls, graph, node, **kw): - mul_node = graph.make_node( - 'Mul', inputs=[node.input('X', 0), node.input('Y', 0)]) - one = graph.make_node( - 'Constant', - dtype=dtypes.ONNX.INT64, - value=[len(node.input_shape('X', 0)) - 1]) - graph.make_node( - 'ReduceSum', inputs=[mul_node, one], outputs=node.output('Out')) - - -@op_mapper('dist') -class Dist(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - sub_node = graph.make_node( - 'Sub', inputs=[node.input('X', 0), node.input('Y', 0)]) - abs_node = graph.make_node('Abs', inputs=sub_node) - if node.attr('p') == 0: - assert graph.opset_version >= 9, "When p is 0, onnx opset should be (onnx_opset>=9)." - sign_node = graph.make_node('Sign', inputs=abs_node) - sum_node = graph.make_node( - 'ReduceSum', inputs=sign_node, keepdims=0) - mapper_helper.unsqueeze_helper(graph, sum_node, [0], - node.output('Out')) - elif node.attr('p') == float('inf'): - max_node = graph.make_node('ReduceMax', inputs=abs_node, keepdims=0) - mapper_helper.unsqueeze_helper(graph, max_node, [0], - node.output('Out')) - elif node.attr('p') == float('-inf'): - min_node = graph.make_node('ReduceMin', inputs=abs_node, keepdims=0) - mapper_helper.unsqueeze_helper(graph, min_node, [0], - node.output('Out')) - else: - x_dtype = node.input_dtype('X', 0) - p = graph.make_node( - 'Constant', - dtype=dtypes.DTYPE_PADDLE_ONNX_MAP[x_dtype], - value=node.attr('p')) - pow_node = graph.make_node( - 'Pow', - inputs=[abs_node, p], ) - sum_node = graph.make_node('ReduceSum', inputs=pow_node, keepdims=0) - sum_node = mapper_helper.unsqueeze_helper(graph, sum_node, [0]) - p_1 = graph.make_node('Reciprocal', inputs=p) - graph.make_node( - 'Pow', inputs=[sum_node, p_1], outputs=node.output('Out')) - - -@op_mapper('round') -class Round(): - support_opset_version_range = (11, 15) - - @classmethod - def opset_11(cls, graph, node, **kw): - graph.make_node( - 'Round', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('rsqrt') -class Rsqrt(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_6(cls, graph, node, **kw): - sqrt_node = graph.make_node('Sqrt', inputs=node.input('X')) - graph.make_node( - 'Reciprocal', inputs=sqrt_node, outputs=node.output('Out')) - - -@op_mapper('sign') -class Sign(): - support_opset_version_range = (9, 15) - - @classmethod - def opset_9(cls, graph, node, **kw): - graph.make_node( - 'Sign', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('scale') -class Scale(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - scale = node.attr('scale') - bias = node.attr('bias') - if len(node.input('ScaleTensor')) == 0 and np.fabs( - scale - 1.0) < 1e-06 and np.fabs(bias - 0.0) < 1e-06: - graph.make_node( - 'Identity', inputs=node.input('X'), outputs=node.output('Out')) - else: - input_dtype = dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)] - if input_dtype in [ - dtypes.ONNX.INT16, dtypes.ONNX.INT32, dtypes.ONNX.INT64 - ]: - outputs = None - data_type = dtypes.ONNX.FLOAT - cast_node = graph.make_node( - 'Cast', inputs=node.input('X'), attrs={'to': data_type}) - else: - outputs = node.output('Out') - data_type = input_dtype - cast_node = node.input('X')[0] - - if len(node.input('ScaleTensor')) > 0: - scale_node = node.input('ScaleTensor')[0] - scale_type = dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype( - 'ScaleTensor', 0)] - if scale_type != data_type: - scale_node = graph.make_node( - 'Cast', inputs=[scale_node], attrs={'to': data_type}) - else: - scale_node = graph.make_node( - 'Constant', attrs={'dtype': data_type, - 'value': [scale]}) - bias_node = graph.make_node( - 'Constant', attrs={'dtype': data_type, - 'value': [bias]}) - - if node.attr('bias_after_scale'): - node1 = graph.make_node('Mul', inputs=[cast_node, scale_node]) - node2 = graph.make_node( - 'Add', inputs=[node1, bias_node], outputs=outputs) - else: - node1 = graph.make_node('Add', inputs=[cast_node, bias_node]) - node2 = graph.make_node( - 'Mul', inputs=[node1, scale_node], outputs=outputs) - - if input_dtype in [ - dtypes.ONNX.INT16, dtypes.ONNX.INT32, dtypes.ONNX.INT64 - ]: - cast_node = graph.make_node( - 'Cast', - inputs=node2, - outputs=node.output('Out'), - attrs={'to': input_dtype}) - - -@op_mapper('softmax') -class Softmax(): - support_opset_version_range = (1, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - axis = node.attr('axis') - shape = node.output_shape('Out', 0) - if axis is None: - axis = -1 - if axis < 0: - axis += len(shape) - if axis == len(shape) - 1: - node = graph.make_node( - 'Softmax', - inputs=node.input('X'), - outputs=node.output('Out'), - attrs={'axis': axis}) - else: - perm = [i for i in range(len(shape))] - perm[-1] = axis - perm[axis] = len(shape) - 1 - transpose_node = graph.make_node( - 'Transpose', inputs=node.input('X'), attrs={'perm': perm}) - softmax_node = graph.make_node( - 'Softmax', inputs=[transpose_node], axis=-1) - transpose_node1 = graph.make_node( - 'Transpose', - inputs=[softmax_node], - outputs=node.output('Out'), - attrs={'perm': perm}) - - @classmethod - def opset_13(cls, graph, node, **kw): - graph.make_node( - 'Softmax', - inputs=node.input('X'), - axis=node.attr('axis'), - outputs=node.output('Out')) - - -@op_mapper('unfold') -class Unfold(): - support_opset_version_range = (11, 15) - - @classmethod - def opset_11(cls, graph, node, **kw): - - strides = node.attr('strides') - stride_h = strides[0] - stride_w = strides[1] - - paddings = node.attr('paddings') - padding_h_1 = paddings[0] - padding_w_1 = paddings[1] - padding_h_2 = paddings[2] - padding_w_2 = paddings[3] - - dilations = node.attr('dilations') - dilation_h = dilations[0] - dilation_w = dilations[1] - - kernel_sizes = node.attr('kernel_sizes') - kernel_h = kernel_sizes[0] - kernel_w = kernel_sizes[1] - - input_w = mapper_helper.shape_helper(graph, node.input('X', 0), 3) - blocks_row_indices_node = cls._get_im2col_indices_along_dim( - graph, node, 2, kernel_h, dilation_h, padding_h_1, padding_h_2, - stride_h) - blocks_col_indices_node = cls._get_im2col_indices_along_dim( - graph, node, 3, kernel_w, dilation_w, padding_w_1, padding_w_2, - stride_w) - - output_shape = cls._get_im2col_output_shape(graph, node, kernel_h, - kernel_w) - padded_input = cls._get_im2col_padded_input( - graph, node, padding_h_1, padding_h_2, padding_w_1, padding_w_2) - - output = graph.make_node( - 'Gather', inputs=[padded_input, blocks_row_indices_node], axis=2) - - output = graph.make_node( - 'Gather', inputs=[output, blocks_col_indices_node], axis=4) - output = graph.make_node( - 'Transpose', inputs=[output], perm=[0, 1, 2, 4, 3, 5]) - - graph.make_node( - 'Reshape', inputs=[output, output_shape], outputs=node.output('Y')) - - @classmethod - def _get_im2col_indices_along_dim(cls, graph, node, index, kernel_size_d, - dilation_d, padding_d_1, padding_d_2, - stride_d): - input_shape = node.input_shape('X', 0) - if input_shape[index] == -1: - input_d_node = mapper_helper.shape_helper(graph, - node.input('X', 0), index) - - padding_d_node = graph.make_node( - 'Constant', - dtype=dtypes.ONNX.INT64, - value=[padding_d_1 + padding_d_2]) - blocks_d_node = graph.make_node( - 'Add', inputs=[input_d_node, padding_d_node]) - - dilation_kernel_size_node = graph.make_node( - 'Constant', - dtype=dtypes.ONNX.INT64, - value=[dilation_d * (kernel_size_d - 1)]) - blocks_d_node = graph.make_node( - 'Sub', inputs=[blocks_d_node, dilation_kernel_size_node]) - - zero_node = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[0]) - stride_node = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[stride_d]) - blocks_d_indices_node = graph.make_node( - 'Range', inputs=[zero_node, blocks_d_node, stride_node]) - else: - end = input_shape[ - index] + padding_d_1 + padding_d_2 - dilation_d * (kernel_size_d - - 1) - stride = stride_d - blocks_d_indices = np.arange(0, end, stride) - blocks_d_indices_node = graph.make_node( - 'Constant', - dtype=dtypes.ONNX.INT64, - value=blocks_d_indices.flatten().tolist()) - - kernel_grid = np.arange(0, kernel_size_d * dilation_d, dilation_d) - kernel_grid_node = graph.make_node( - 'Constant', - dtype=dtypes.ONNX.INT64, - value=kernel_grid.flatten().tolist()) - - shape_node = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[-1, 1]) - kernel_mask_node = graph.make_node( - 'Reshape', inputs=[kernel_grid_node, shape_node]) - - block_mask_node = graph.make_node( - 'Add', inputs=[blocks_d_indices_node, kernel_mask_node]) - return block_mask_node - - @classmethod - def _get_im2col_output_shape(cls, graph, node, kernel_h, kernel_w): - batch_dim = mapper_helper.shape_helper(graph, node.input('X', 0), 0) - channel_dim = mapper_helper.shape_helper(graph, node.input('X', 0), 1) - - constant_node = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[kernel_h * kernel_w]) - channel_unfolded = graph.make_node( - 'Mul', inputs=[channel_dim, constant_node]) - - concat_const_node = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[-1]) - result_node = graph.make_node( - 'Concat', - inputs=[batch_dim, channel_unfolded, concat_const_node], - axis=0) - - return result_node - - @classmethod - def _get_im2col_padded_input(cls, graph, node, padding_h_1, padding_h_2, - padding_w_1, padding_w_2): - pad_const_node = graph.make_node( - 'Constant', - dtype=dtypes.ONNX.INT64, - value=[ - 0, 0, padding_h_1, padding_w_1, 0, 0, padding_h_2, padding_w_2 - ]) - result_node = graph.make_node( - 'Pad', inputs=[node.input('X', 0), pad_const_node]) - return result_node - - -@op_mapper('softmax_with_cross_entropy') -class SoftmaxCrossEntropyLoss(): - support_opset_version_range = (12, 15) - - @classmethod - def opset_12(cls, graph, node, **kw): - if node.attr('soft_label'): - raise Exception( - "SoftmaxCrossEntropyLoss in onnx not support soft label.") - scores = node.input('Logits', 0) - labels = node.input('Label', 0) - # Whether return_softmax is True or False, the model will have two outputs - outputs = [node.output('Loss', 0), node.output('Softmax', 0)] - - shape = node.input_shape('Logits', 0) - if len(shape) < 2: - raise Exception( - "SoftmaxCrossEntropyLoss in onnx not support 1D logits.") - axis = node.attr('axis') - if axis < 0: - axis += len(shape) - if axis == 1: - squeeze_node = mapper_helper.squeeze_helper(graph, labels, [axis]) - loss_node, softmax_node = graph.make_node( - 'SoftmaxCrossEntropyLoss', - inputs=[scores, squeeze_node], - outputs=2, - ignore_index=node.attr('ignore_index'), - reduction='none') - loss_node = mapper_helper.unsqueeze_helper(graph, loss_node, - [axis], outputs[0]) - # onnx output is log(softmax), but paddle output is softmax - graph.make_node('Exp', inputs=[softmax_node], outputs=outputs[1]) - else: - perm = [i for i in range(len(shape))] - perm[1] = axis - perm[axis] = 1 - transpose_scores = graph.make_node( - 'Transpose', inputs=[scores], perm=perm) - transpose_labels = graph.make_node( - 'Transpose', inputs=[labels], perm=perm) - squeeze_labels = mapper_helper.squeeze_helper( - graph, transpose_labels, [1]) - - loss_node, softmax_node = graph.make_node( - 'SoftmaxCrossEntropyLoss', - inputs=[transpose_scores, squeeze_labels], - ignore_index=node.attr('ignore_index'), - outputs=2, - reduction='none') - output_node = mapper_helper.unsqueeze_helper(graph, loss_node, [1]) - graph.make_node( - 'Transpose', inputs=output_node, outputs=outputs[0], perm=perm) - softmax_node = graph.make_node( - 'Transpose', inputs=softmax_node, perm=perm) - # onnx output is log(softmax), but paddle output is softmax - graph.make_node('Exp', inputs=[softmax_node], outputs=outputs[1]) diff --git a/paddle2onnx/legacy/op_mapper/nn.py b/paddle2onnx/legacy/op_mapper/nn.py deleted file mode 100755 index 37e2f1fabca..00000000000 --- a/paddle2onnx/legacy/op_mapper/nn.py +++ /dev/null @@ -1,952 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import numpy as np -import math -import collections -from paddle2onnx.legacy.constant import dtypes -from paddle2onnx.legacy.op_mapper import OpMapper as op_mapper -from paddle2onnx.legacy.op_mapper import mapper_helper -from paddle2onnx import utils -import paddle - - -@op_mapper(['conv2d', 'depthwise_conv2d', 'conv3d']) -class Conv(): - support_opset_version_range = (1, 12) - - @classmethod - def opset_1(cls, graph, node, **kw): - kernel_shape = node.input_shape('Filter', 0) - dilations = node.attr('dilations') - kernel_shape = kernel_shape[2:] - strides = node.attr('strides') - group = node.attr('groups') - pads = node.attr('paddings') - assert node.attrs['data_format'] == 'NCHW' or node.attrs['data_format'] == 'NCDHW' or node.attrs['data_format'] == "AnyLayout", \ - "The conv data format should be 'NCHW' or 'NCDHW', but received data format " \ - "is %s." % node.attrs['data_format'] - # onnx padding is [x1_begin, x2_begin...x1_end, x2_end, ...] - if len(pads) == 2 or len(pads) == 3: - pads = pads + pads - elif len(pads) == 4: - pads = [pads[i] for i in [0, 2, 1, 3]] - elif len(pads) == 6: - pads = [pads[i] for i in [0, 2, 4, 1, 3, 5]] - attrs = { - 'dilations': dilations, - 'kernel_shape': kernel_shape, - 'strides': strides, - 'group': group - } - auto_pad = node.attr('padding_algorithm') - if auto_pad == 'SAME': - attrs['auto_pad'] = 'SAME_UPPER' - elif auto_pad == 'VALID': - attrs['auto_pad'] = 'VALID' - else: - attrs['pads'] = pads - graph.make_node( - 'Conv', - inputs=node.input('Input') + node.input('Filter'), - outputs=node.output('Output'), - attrs=attrs) - - -@op_mapper( - ['conv2d_transpose', 'depthwise_conv2d_transpose', 'conv3d_transpose']) -class ConvTranspose(): - support_opset_version_range = (1, 12) - - @classmethod - def opset_1(cls, graph, node, **kw): - output_padding = node.attr('output_padding') - kernel_shape = node.input_shape('Filter', 0) - dilations = node.attr('dilations') - kernel_shape = kernel_shape[2:] - strides = node.attr('strides') - group = node.attr('groups') - pads = node.attr('paddings') - assert node.attrs['data_format'] == 'NCHW' or node.attrs['data_format'] == 'NCDHW', \ - "The conv data format should be 'NCHW' or 'NCDHW', but received data format " \ - "is %s." % node.attrs['data_format'] - - if len(pads) == 2 or len(pads) == 3: - pads = pads + pads - elif len(pads) == 4: - pads = [pads[i] for i in [0, 2, 1, 3]] - elif len(pads) == 6: - pads = [pads[i] for i in [0, 2, 4, 1, 3, 5]] - - attrs = { - 'dilations': dilations, - 'kernel_shape': kernel_shape, - 'strides': strides, - 'group': group - } - auto_pad = node.attr('padding_algorithm') - if auto_pad == 'SAME': - attrs['auto_pad'] = 'SAME_UPPER' - elif auto_pad == 'VALID': - attrs['auto_pad'] = 'VALID' - else: - attrs['pads'] = pads - if output_padding and len(output_padding) > 0: - attrs['output_padding'] = output_padding - graph.make_node( - 'ConvTranspose', - inputs=node.input('Input') + node.input('Filter'), - outputs=node.output('Output'), - attrs=attrs) - - -@op_mapper('pool2d') -class Pool(): - support_opset_version_range = (1, 12) - pool_type = { - 'max': ('MaxPool', 'GlobalMaxPool'), - 'avg': ('AveragePool', 'GlobalAveragePool') - } - - @classmethod - def is_same_span(cls, in_size, out_size): - spans = [] - for i in range(out_size): - start = math.floor(i * (in_size / out_size)) - end = math.ceil((i + 1) * (in_size / out_size)) - spans.append(end - start) - if len(set(spans)) == 1: - return True - return False - - @classmethod - def opset_1(cls, graph, node, **kw): - assert node.attrs['data_format'] == 'NCHW' or node.attrs['data_format'] == "AnyLayout", \ - "The conv data format should be 'NCHW', but received data format " \ - "is %s." % node.attrs['data_format'] - x_dtype = node.input_dtype('X', 0) - need_dtype_convert = False - input_name = node.input('X', 0) - if x_dtype != paddle.float32: - need_dtype_convert = True - input_name = graph.make_node( - 'Cast', inputs=node.input('X'), to=dtypes.ONNX.FLOAT) - - if node.attr('global_pooling') or (node.attr('adaptive') and - node.attr('ksize') == [1, 1]): - if need_dtype_convert: - onnx_node = graph.make_node( - cls.pool_type[node.attr('pooling_type')][1], - inputs=[input_name]) - graph.make_node( - 'Cast', - inputs=[onnx_node], - outputs=node.output('Out'), - to=dtypes.ONNX.DOUBLE) - else: - onnx_node = graph.make_node( - cls.pool_type[node.attr('pooling_type')][1], - inputs=[input_name], - outputs=node.output('Out')) - elif node.attr('adaptive'): - # if pool is adaptive, check if input shape of pool is fixed. - if node.input_shape('X', 0)[2:].count(-1) > 0: - raise Exception( - "Converting this model to ONNX need with static input shape," \ - " please fix input shape of this model, see doc Q2 in" \ - " https://github.com/PaddlePaddle/paddle2onnx/blob/develop/docs/en/FAQ.md." - ) - input_h, input_w = node.input_shape('X', 0)[2:] - output_h, output_w = node.output_shape('Out', 0)[2:] - stride_h = int(input_h / output_h) - stride_w = int(input_w / output_w) - - kernel_h = input_h - (output_h - 1) * stride_h - kernel_w = input_w - (output_w - 1) * stride_w - - #check if kernel_size is fixed. - if not cls.is_same_span(input_h, output_h) or not cls.is_same_span( - input_w, output_w): - raise Exception( - "Cannot convert adaptive pool with input_size: {}, output_size: {}" - .format( - node.input_shape('X', 0), node.output_shape('Out', 0))) - else: - attrs = { - 'kernel_shape': (kernel_h, kernel_w), - 'strides': (stride_h, stride_w), - } - if node.attr('ceil_mode') and graph.opset_version < 10: - raise Exception( - "Cannot convert pool with ceil_model == True to ONNX Opset version < 10." - ) - elif graph.opset_version > 10: - attrs['ceil_mode'] = node.attr('ceil_mode') - auto_pad = node.attr('padding_algorithm') - if auto_pad == 'SAME': - attrs['auto_pad'] = 'SAME_UPPER' - elif auto_pad == 'VALID': - attrs['auto_pad'] = 'VALID' - if node.attr('pooling_type') == 'avg': - attrs['count_include_pad'] = not node.attr('exclusive') - if need_dtype_convert: - onnx_node = graph.make_node( - cls.pool_type[node.attr('pooling_type')][0], - inputs=[input_name], - attrs=attrs) - graph.make_node( - 'Cast', - inputs=[onnx_node], - outputs=node.output('Out'), - to=dtypes.ONNX.DOUBLE) - else: - onnx_node = graph.make_node( - cls.pool_type[node.attr('pooling_type')][0], - inputs=[input_name], - outputs=node.output('Out'), - attrs=attrs) - else: - input_shape = node.input_shape('X', 0) - k_size = node.attr('ksize') - pads = node.attr('paddings') - strides = node.attr('strides') - - if len(pads) == 2: - pads = pads + pads - elif len(pads) == 4: - pads = [pads[i] for i in [0, 2, 1, 3]] - - if input_shape[2] > 0 and input_shape[2] + pads[0] < k_size[0]: - k_size[0] = input_shape[2] + pads[0] - if input_shape[3] > 0 and input_shape[3] + pads[1] < k_size[1]: - k_size[1] = input_shape[3] + pads[1] - - input_x = [input_name] - if max(k_size) <= max(pads): - onnx_paddings = [0, 0, pads[0], pads[1], 0, 0, pads[2], pads[3]] - attrs_pad = {'mode': 'constant', } - if graph.opset_version >= 11: - pads_node = graph.make_node( - 'Constant', - attrs={ - 'dtype': dtypes.ONNX.INT64, - 'value': onnx_paddings - }) - value_node = graph.make_node( - 'Constant', - attrs={'dtype': dtypes.ONNX.FLOAT, - 'value': 0.0}) - input_x = input_x + [pads_node, value_node] - else: - attrs_pad['pads'] = onnx_paddings - attrs_pad['value'] = 0.0 - input_x = graph.make_node( - 'Pad', inputs=input_x, attrs=attrs_pad) - pads = [0, 0, 0, 0] - - attrs = { - 'kernel_shape': k_size, - 'strides': strides, - } - auto_pad = node.attr('padding_algorithm') - if auto_pad == 'SAME': - attrs['auto_pad'] = 'SAME_UPPER' - elif auto_pad == 'VALID': - attrs['auto_pad'] = 'VALID' - else: - attrs['pads'] = pads - if node.attr('ceil_mode') and graph.opset_version < 10: - raise Exception( - "Cannot convert pool with ceil_model == True to ONNX Opset version < 10" - ) - elif graph.opset_version >= 10: - attrs['ceil_mode'] = node.attr('ceil_mode') - - if node.attr('pooling_type') == 'avg': - attrs['count_include_pad'] = not node.attr('exclusive') - if need_dtype_convert: - onnx_node = graph.make_node( - cls.pool_type[node.attr('pooling_type')][0], - inputs=input_x, - attrs=attrs) - graph.make_node( - 'Cast', - inputs=[onnx_node], - outputs=node.output('Out'), - to=dtypes.ONNX.DOUBLE) - else: - onnx_node = graph.make_node( - cls.pool_type[node.attr('pooling_type')][0], - inputs=input_x, - outputs=node.output('Out'), - attrs=attrs) - - -@op_mapper('pool3d') -class Pool3D(): - support_opset_version_range = (1, 12) - pool_type = { - 'max': ('MaxPool', 'GlobalMaxPool'), - 'avg': ('AveragePool', 'GlobalAveragePool') - } - - @classmethod - def is_same_span(cls, in_size, out_size): - spans = [] - for i in range(out_size): - start = math.floor(i * (in_size / out_size)) - end = math.ceil((i + 1) * (in_size / out_size)) - spans.append(end - start) - if len(set(spans)) == 1: - return True - return False - - @classmethod - def opset_1(cls, graph, node, **kw): - assert node.attrs['data_format'] == 'NCDHW' or node.attrs['data_format'] == "AnyLayout", \ - "The conv data format should be 'NCDHW', but received data format " \ - "is %s." % node.attrs['data_format'] - - if node.attr('global_pooling') or (node.attr('adaptive') and - node.attr('ksize') == [1, 1, 1]): - onnx_node = graph.make_node( - cls.pool_type[node.attr('pooling_type')][1], - inputs=node.input('X'), - outputs=node.output('Out')) - elif node.attr('adaptive'): - # if pool is adaptive, check if input shape of pool is fixed. - if node.input_shape('X', 0)[2:].count(-1) > 0: - raise Exception( - "Converting this model to ONNX need with static input shape," \ - " please fix input shape of this model, see doc Q2 in" \ - " https://github.com/PaddlePaddle/paddle2onnx/blob/develop/docs/en/FAQ.md." - ) - input_d, input_h, input_w = node.input_shape('X', 0)[2:] - output_d, output_h, output_w = node.output_shape('Out', 0)[2:] - stride_d = int(input_d / output_d) - stride_h = int(input_h / output_h) - stride_w = int(input_w / output_w) - - kernel_d = input_d - (output_d - 1) * stride_d - kernel_h = input_h - (output_h - 1) * stride_h - kernel_w = input_w - (output_w - 1) * stride_w - - #check if kernel_size is fixed. - if not cls.is_same_span(input_h, output_h) or not cls.is_same_span( - input_w, output_w) or not cls.is_same_span(input_d, - output_d): - raise Exception( - "Cannot convert adaptive pool with input_size: {}, output_size: {}" - .format( - node.input_shape('X', 0), node.output_shape('Out', 0))) - else: - attrs = { - 'kernel_shape': (kernel_d, kernel_h, kernel_w), - 'strides': (stride_d, stride_h, stride_w), - } - if node.attr('ceil_mode') and graph.opset_version < 10: - raise Exception( - "Cannot convert pool with ceil_model == True to ONNX Opset version < 10." - ) - elif graph.opset_version > 10: - attrs['ceil_mode'] = node.attr('ceil_mode') - auto_pad = node.attr('padding_algorithm') - if auto_pad == 'SAME': - attrs['auto_pad'] = 'SAME_UPPER' - elif auto_pad == 'VALID': - attrs['auto_pad'] = 'VALID' - if node.attr('pooling_type') == 'avg': - attrs['count_include_pad'] = not node.attr('exclusive') - onnx_node = graph.make_node( - cls.pool_type[node.attr('pooling_type')][0], - inputs=node.input('X'), - outputs=node.output('Out'), - attrs=attrs) - else: - input_shape = node.input_shape('X', 0) - k_size = node.attr('ksize') - paddings = node.attr('paddings') - if input_shape[2] > 0 and input_shape[2] + paddings[0] < k_size[0]: - k_size[0] = input_shape[2] + paddings[0] - if input_shape[3] > 0 and input_shape[3] + paddings[1] < k_size[1]: - k_size[1] = input_shape[3] + paddings[1] - if input_shape[4] > 0 and input_shape[4] + paddings[2] < k_size[2]: - k_size[2] = input_shape[4] + paddings[2] - attrs = { - 'kernel_shape': k_size, - 'strides': node.attr('strides'), - 'pads': node.attr('paddings') + node.attr('paddings'), - } - if node.attr('ceil_mode') and graph.opset_version < 10: - raise Exception( - "Cannot convert pool with ceil_model == True to ONNX Opset version < 10" - ) - elif graph.opset_version >= 10: - attrs['ceil_mode'] = node.attr('ceil_mode') - - if node.attr('pooling_type') == 'avg': - attrs['count_include_pad'] = not node.attr('exclusive') - onnx_node = graph.make_node( - cls.pool_type[node.attr('pooling_type')][0], - inputs=node.input('X'), - outputs=node.output('Out'), - attrs=attrs) - - -@op_mapper('elu') -class ELU(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - node = graph.make_node( - 'Elu', - inputs=node.input('X'), - outputs=node.output('Out'), - alpha=node.attr('alpha')) - - -@op_mapper('softsign') -class SoftSign(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - graph.make_node( - 'Softsign', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('hard_shrink') -class Hardshrink(): - support_opset_version_range = (9, 15) - - @classmethod - def opset_9(cls, graph, node, **kw): - node = graph.make_node( - 'Shrink', - inputs=node.input('X'), - outputs=node.output('Out'), - lambd=node.attr('threshold')) - - -@op_mapper('logsigmoid') -class LogSigmoid(): - support_opset_version_range = (1, 12) - - @classmethod - def opset_1(cls, graph, node, **kw): - sigmoid_node = graph.make_node('Sigmoid', inputs=node.input('X')) - graph.make_node('Log', inputs=sigmoid_node, outputs=node.output('Out')) - - -@op_mapper('norm') -class Norm(): - support_opset_version_range = (1, 12) - - @classmethod - def opset_1(cls, graph, node, **kw): - node = graph.make_node( - 'LpNormalization', - inputs=node.input('X'), - outputs=node.output('Out'), - axis=node.attr('axis')) - - -@op_mapper('softshrink') -class SoftShrink(): - support_opset_version_range = (9, 15) - - @classmethod - def opset_9(cls, graph, node, **kw): - graph.make_node( - 'Shrink', - inputs=node.input('X'), - bias=node.attr('lambda'), - lambd=node.attr('lambda'), - outputs=node.output('Out')) - - -@op_mapper('tanh_shrink') -class TanhShrink(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - tanh_node = graph.make_node( - 'Tanh', - inputs=node.input('X', 0), ) - graph.make_node( - 'Sub', - inputs=[node.input('X', 0), tanh_node], - outputs=node.output('Out')) - - -@op_mapper('log_softmax') -class LogSoftmax(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - axis = node.attr('axis') - shape = node.output_shape('Out', 0) - if axis is None: - axis = -1 - if axis < 0: - axis += len(shape) - if axis == len(shape) - 1: - node = graph.make_node( - 'LogSoftmax', - inputs=node.input('X'), - outputs=node.output('Out'), - attrs={'axis': axis}) - else: - perm = [i for i in range(len(shape))] - perm[-1] = axis - perm[axis] = len(shape) - 1 - transpose_node = graph.make_node( - 'Transpose', inputs=node.input('X'), attrs={'perm': perm}) - softmax_node = graph.make_node( - 'LogSoftmax', inputs=[transpose_node], axis=-1) - transpose_node1 = graph.make_node( - 'Transpose', - inputs=[softmax_node], - outputs=node.output('Out'), - attrs={'perm': perm}) - - @classmethod - def opset_13(cls, graph, node, **kw): - graph.make_node( - 'LogSoftmax', - inputs=node.input('X'), - axis=node.attr('axis'), - outputs=node.output('Out')) - - -@op_mapper('layer_norm') -class LayerNorm(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - ipt = node.input('X', 0) - ipt_dims = len(node.input_shape('X', 0)) - normalized_shape = node.attr('begin_norm_axis') - axes = None - if isinstance(normalized_shape, collections.Iterable): - axes = [-i for i in range(len(normalized_shape), 0, -1)] - else: - axes = [i for i in range(normalized_shape, ipt_dims)] - dtype = node.block.vars[node.input('X', 0)].dtype - dtype = dtypes.DTYPE_PADDLE_ONNX_MAP[dtype] - epsilon = graph.make_node( - 'Constant', dtype=dtype, value=node.attr('epsilon')) - two = graph.make_node('Constant', dtype=dtype, value=2.0) - mean = graph.make_node("ReduceMean", inputs=[ipt], axes=axes) - numerator = graph.make_node("Sub", inputs=[ipt, mean]) - pow_num = graph.make_node("Pow", inputs=[numerator, two]) - variance = graph.make_node("ReduceMean", inputs=[pow_num], axes=axes) - add_eps = graph.make_node("Add", inputs=[variance, epsilon]) - denominator = graph.make_node("Sqrt", inputs=[add_eps]) - - ipt_shape = graph.make_node("Shape", inputs=[ipt]) - weight_shape = mapper_helper.slice_helper( - graph, ipt_shape, [0], [ipt_dims - len(axes)], [ipt_dims]) - if 'Bias' in node.inputs and 'Scale' in node.inputs and len( - node.input('Scale')) > 0 and len(node.input('Bias')) > 0: - if normalized_shape == ipt_dims - 1: - shape_const = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[-1]) - scale = graph.make_node( - "Reshape", inputs=[node.input('Scale', 0), shape_const]) - bias = graph.make_node( - "Reshape", inputs=[node.input('Bias', 0), shape_const]) - else: - scale = graph.make_node( - "Reshape", inputs=[node.input('Scale', 0), weight_shape]) - bias = graph.make_node( - "Reshape", inputs=[node.input('Bias', 0), weight_shape]) - layer_norm = graph.make_node("Div", inputs=[numerator, denominator]) - layer_norm = graph.make_node("Mul", inputs=[layer_norm, scale]) - graph.make_node( - "Add", inputs=[layer_norm, bias], outputs=node.output('Y')) - elif 'Bias' in node.inputs and len(node.input('Bias')) > 0: - if normalized_shape == ipt_dims - 1: - shape_const = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[-1]) - bias = graph.make_node( - "Reshape", inputs=[node.input('Bias', 0), shape_const]) - else: - bias = graph.make_node( - "Reshape", inputs=[node.input('Bias', 0), weight_shape]) - layer_norm = graph.make_node("Div", inputs=[numerator, denominator]) - graph.make_node( - "Add", inputs=[layer_norm, bias], outputs=node.output('Y')) - elif 'Scale' in node.inputs and len(node.input('Scale')) > 0: - if normalized_shape == ipt_dims - 1: - shape_const = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[-1]) - scale = graph.make_node( - "Reshape", inputs=[node.input('Scale', 0), shape_const]) - else: - scale = graph.make_node( - "Reshape", inputs=[node.input('Scale', 0), weight_shape]) - layer_norm = graph.make_node("Div", inputs=[numerator, denominator]) - graph.make_node( - "Mul", inputs=[layer_norm, scale], outputs=node.output('Y')) - else: - layer_norm = graph.make_node( - "Div", - inputs=[numerator, denominator], - outputs=node.output('Y')) - - -@op_mapper('batch_norm') -class BatchNorm(): - support_opset_version_range = (7, 15) - - @classmethod - def make_attrs_and_inputs(cls, graph, node, **kw): - onnx_attr = { - 'epsilon': node.attr('epsilon'), - 'momentum': node.attr('momentum') - } - inputs = node.input('X') + node.input('Scale') + node.input( - 'Bias') + node.input('Mean') + node.input('Variance') - return onnx_attr, inputs - - @classmethod - def opset_9(cls, graph, node, **kw): - onnx_attr, inputs = cls.make_attrs_and_inputs(graph, node, **kw) - onnx_node = graph.make_node( - 'BatchNormalization', - inputs=inputs, - outputs=node.output('Y'), - **onnx_attr) - - @classmethod - def opset_7(cls, graph, node, **kw): - onnx_attr, inputs = cls.make_attrs_and_inputs(graph, node, **kw) - onnx_attr['spatial'] = 1 - onnx_node = graph.make_node( - 'BatchNormalization', - inputs=inputs, - outputs=node.output('Y'), - **onnx_attr) - - -@op_mapper('group_norm') -class GroupNorm(): - support_opset_version_range = (6, 15) - - @classmethod - def opset_6(cls, graph, node, **kw): - num_groups = node.attr('groups') - epsilon = node.attr('epsilon') - ipt = node.input('X')[0] - - ipt_shape = node.input_shape('X', 0) - assert len( - ipt_shape) == 4, "Only support 4D-Tensor as input for GroupNorm" - - dtype = node.block.vars[node.input('X', 0)].dtype - dtype = dtypes.DTYPE_PADDLE_ONNX_MAP[dtype] - - shape = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[0, num_groups, -1]) - reshape_input = graph.make_node('Reshape', inputs=[ipt, shape]) - scale_ = graph.make_node( - 'Constant', dtype=dtype, value=[1.0] * num_groups) - bias_ = graph.make_node( - 'Constant', dtype=dtype, value=[0.0] * num_groups) - reshaped_output = graph.make_node( - 'InstanceNormalization', - inputs=[reshape_input, scale_, bias_], - epsilon=epsilon) - origin_shape = graph.make_node('Shape', inputs=[ipt]) - - if len(node.input('Scale')) > 0 and len(node.input('Bias')) > 0: - output = graph.make_node( - 'Reshape', inputs=[reshaped_output, origin_shape]) - unsqueezed_scale = mapper_helper.unsqueeze_helper( - graph, node.input('Scale', 0), [1, 2]) - unsqueezed_bias = mapper_helper.unsqueeze_helper( - graph, node.input('Bias', 0), [1, 2]) - part0 = graph.make_node('Mul', inputs=[output, unsqueezed_scale]) - graph.make_node( - 'Add', - inputs=[part0, unsqueezed_bias], - outputs=node.output('Y')) - else: - output = graph.make_node( - 'Reshape', - inputs=[reshaped_output, origin_shape], - outputs=node.output('Y')) - - -@op_mapper('instance_norm') -class InstanceNorm(): - support_opset_version_range = (6, 15) - - @classmethod - def opset_6(cls, graph, node, **kw): - onnx_attr = {'epsilon': node.attr('epsilon'), } - num_groups = node.block.vars[node.input('X')[0]].shape[1] - - dtype = node.block.vars[node.input('X', 0)].dtype - dtype = dtypes.DTYPE_PADDLE_ONNX_MAP[dtype] - - if len(node.input('Scale')) == 0: - scale_ = graph.make_node( - 'Constant', dtype=dtype, value=[1.0] * num_groups) - else: - scale_ = node.input('Scale')[0] - if len(node.input('Bias')) == 0: - bias_ = graph.make_node( - 'Constant', dtype=dtype, value=[0.0] * num_groups) - else: - bias_ = node.input('Bias')[0] - - inputs = node.input('X') + [scale_] + [bias_] - onnx_node = graph.make_node( - 'InstanceNormalization', - inputs=inputs, - outputs=node.output('Y'), - **onnx_attr) - - -@op_mapper('dropout') -class Dropout(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - dropout_mode = node.attr('dropout_implementation') - dropout_prob = node.attr('dropout_prob') - if dropout_mode == 'upscale_in_train': - onnx_node = graph.make_node( - 'Identity', inputs=node.input('X'), outputs=node.output('Out')) - elif dropout_mode == 'downgrade_in_infer': - scale_node = graph.make_node( - 'Constant', - attrs={'dtype': dtypes.ONNX.FLOAT, - 'value': 1 - dropout_prob}) - graph.make_node( - "Mul", - inputs=[node.input('X')[0], scale_node], - outputs=node.output('Out')) - else: - raise Exception("Unexpected situation happend") - - -@op_mapper('roi_align') -class RoiAlign(): - support_opset_version_range = (10, 16) - - @classmethod - def opset_10(cls, graph, node, **kw): - if node.attr('aligned') and graph.opset_version < 16: - raise Exception( - 'when aligned is true, onnx opset should be (onnx_opset>= 16)') - rois_shape = graph.make_node('Shape', inputs=[node.input('ROIs', 0)]) - starts = graph.make_node( - 'Constant', attrs={'dtype': dtypes.ONNX.INT64, - 'value': [0]}) - ends = graph.make_node( - 'Constant', attrs={'dtype': dtypes.ONNX.INT64, - 'value': [1]}) - num_rois = graph.make_node('Slice', inputs=[rois_shape, starts, ends]) - zero = graph.make_node( - 'Constant', dims=[1], dtype=dtypes.ONNX.INT64, value=[0]) - batch_indices = graph.make_node('Expand', inputs=[zero, num_rois]) - node = graph.make_node( - 'RoiAlign', - inputs=[node.input('X', 0), node.input('ROIs', 0), batch_indices], - outputs=node.output('Out'), - mode='avg', - output_height=node.attr('pooled_height'), - output_width=node.attr('pooled_width'), - sampling_ratio=node.attr('sampling_ratio'), - spatial_scale=node.attr('spatial_scale')) - - -@op_mapper('rnn') -class RNN(): - support_opset_version_range = (7, 15) - - @classmethod - def make_param_inputs(cls, graph, node, layer, hidden_size, num_layers): - # weight assign order: - # (F_whi F_whh B_whi B_whh)* layer_num + (F_bias_hi F_bias_hh B_bias_hi B_bias_hi)* layer_num - def reform_weights(g, w, n, intervals): - slices = [ - mapper_helper.slice_helper( - g, w, axes=[1], starts=[x * n], ends=[y * n]) - for x, y in intervals - ] - return g.make_node('Concat', slices, axis=1) - - def transform_weight_with_bias(g, weights, n, intervals): - return [reform_weights(g, w, n, intervals) for w in weights] - - if node.attr('mode') == 'LSTM': - reform_permutation = [(0, 1), (3, 4), (1, 3)] - elif node.attr('mode') == 'GRU': - reform_permutation = [(1, 2), (0, 1), (2, 3)] - bidirect_len = 4 if node.attr('is_bidirec') else 2 - all_layer_param_len = len(node.input('WeightList')) - weight_list = node.input('WeightList')[:all_layer_param_len // 2] - bias_list = node.input('WeightList')[all_layer_param_len // 2:] - single_layer_param_len = all_layer_param_len // num_layers - - unsqueeze_weights = [] - layer_weight_list = weight_list[layer * bidirect_len:layer * - bidirect_len + bidirect_len] - layer_bias_list = bias_list[layer * bidirect_len:layer * bidirect_len + - bidirect_len] - param_list = layer_weight_list + layer_bias_list - param_list_len = len(param_list) - for i in range(param_list_len): - weight = mapper_helper.unsqueeze_helper(graph, param_list[i], [0]) - unsqueeze_weights.append(weight) - - input_weights = unsqueeze_weights[0:param_list_len // 2:2] - hidden_weights = unsqueeze_weights[1:param_list_len // 2:2] - - input_weight = graph.make_node('Concat', inputs=input_weights, axis=0) - hidden_weight = graph.make_node('Concat', inputs=hidden_weights, axis=0) - input_bias = unsqueeze_weights[param_list_len // 2:param_list_len:2] - hidden_bias = unsqueeze_weights[param_list_len // 2 + 1:param_list_len: - 2] - - input_bias = graph.make_node('Concat', inputs=input_bias, axis=0) - hidden_bias = graph.make_node('Concat', inputs=hidden_bias, axis=0) - input_weight, hidden_weight, input_bias, hidden_bias = transform_weight_with_bias( - graph, [input_weight, hidden_weight, input_bias, hidden_bias], - hidden_size, reform_permutation) - bias = graph.make_node( - 'Concat', inputs=[input_bias, hidden_bias], axis=1) - return [input_weight, hidden_weight, bias, ''] - - @classmethod - def make_init_param_inputs(cls, graph, node, layer): - if node.attr('mode') == 'LSTM': - all_init_h, all_init_c = node.input('PreState') - bidirect_len = 2 if node.attr('is_bidirec') else 1 - init_h = mapper_helper.slice_helper( - graph, all_init_h, [0], [layer * bidirect_len], - [layer * bidirect_len + bidirect_len]) - init_c = mapper_helper.slice_helper( - graph, all_init_c, [0], [layer * bidirect_len], - [layer * bidirect_len + bidirect_len]) - return [init_h, init_c] - elif node.attr('mode') == 'GRU': - all_init_h = node.input('PreState', 0) - bidirect_len = 2 if node.attr('is_bidirec') else 1 - init_h = mapper_helper.slice_helper( - graph, all_init_h, [0], [layer * bidirect_len], - [layer * bidirect_len + bidirect_len]) - return [init_h] - - @classmethod - def opset_7(cls, graph, node, **kw): - mode = node.attr('mode') - hidden_size = node.attr('hidden_size') - num_layers = node.attr('num_layers') - prev_output = node.input('Input', 0) - if node.attr('mode') == 'LSTM': - for layer in range(num_layers): - param_inputs = cls.make_param_inputs(graph, node, layer, - hidden_size, num_layers) - init_param_inputs = cls.make_init_param_inputs(graph, node, - layer) - if layer + 1 < num_layers: - rnn_outputs = 3 - output_y = None - else: - rnn_outputs = [1] + node.output('State') - output_y = node.output('Out') - prev_output, h_out, c_out = graph.make_node( - node.attr('mode'), - inputs=[prev_output] + param_inputs + init_param_inputs, - outputs=rnn_outputs, - direction='bidirectional' - if node.attr('is_bidirec') else 'forward', - hidden_size=node.attr('hidden_size')) - prev_output = graph.make_node( - 'Transpose', inputs=[prev_output], perm=[0, 2, 1, 3]) - - prev_shape = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[0, 0, -1]) - prev_output = graph.make_node( - 'Reshape', - inputs=[prev_output, prev_shape], - outputs=output_y) - elif node.attr('mode') == 'GRU': - for layer in range(num_layers): - param_inputs = cls.make_param_inputs(graph, node, layer, - hidden_size, num_layers) - init_param_inputs = cls.make_init_param_inputs(graph, node, - layer) - if layer + 1 < num_layers: - rnn_outputs = 2 - output_y = None - else: - rnn_outputs = [1] + node.output('State') - output_y = node.output('Out') - attrs = { - 'direction': 'bidirectional' - if node.attr('is_bidirec') else 'forward', - 'hidden_size': node.attr('hidden_size'), - 'linear_before_reset': 1, - } - prev_output, h_out = graph.make_node( - node.attr('mode'), - inputs=[prev_output] + param_inputs + init_param_inputs, - outputs=rnn_outputs, - attrs=attrs) - prev_output = graph.make_node( - 'Transpose', inputs=[prev_output], perm=[0, 2, 1, 3]) - prev_shape = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[0, 0, -1]) - prev_output = graph.make_node( - 'Reshape', - inputs=[prev_output, prev_shape], - outputs=output_y) - - -@op_mapper('thresholded_relu') -class ThresholdedRelu(): - support_opset_version_range = (10, 15) - - @classmethod - def opset_10(cls, graph, node, **kw): - x_dtype = node.input_dtype('X', 0) - if x_dtype != paddle.float32: - x = graph.make_node( - 'Cast', inputs=node.input('X'), to=dtypes.ONNX.FLOAT) - threshholdedrelu_node = graph.make_node( - 'ThresholdedRelu', inputs=[x], alpha=node.attr('threshold')) - graph.make_node( - 'Cast', - inputs=[threshholdedrelu_node], - outputs=node.output('Out'), - to=dtypes.DTYPE_PADDLE_ONNX_MAP[x_dtype]) - else: - graph.make_node( - 'ThresholdedRelu', - inputs=node.input('X'), - alpha=node.attr('threshold'), - outputs=node.output('Out')) diff --git a/paddle2onnx/legacy/op_mapper/op_mapper.py b/paddle2onnx/legacy/op_mapper/op_mapper.py deleted file mode 100755 index beff194ae45..00000000000 --- a/paddle2onnx/legacy/op_mapper/op_mapper.py +++ /dev/null @@ -1,305 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import inspect -import six -import numpy as np -import paddle -from paddle import fluid -from paddle.fluid import layers - -from paddle2onnx.legacy.graph import graph_helper, PaddleGraph -from paddle2onnx.utils import logging -from paddle2onnx.legacy.constant.op_mapping_status import * - - -REGISTER_CUSTOM_PADDLE_OP = {} - - -def get_max_support_version(versions, opset_version): - max_version = -1 - for vs in sorted(versions): - if vs <= opset_version: - max_version = vs - return max_version - - -def register_op_mapper(paddle_op, mapper_obj): - paddle_op_list = [] - - if isinstance(paddle_op, six.string_types): - paddle_op_list.append(paddle_op) - elif isinstance(paddle_op, list): - paddle_op_list = paddle_op - else: - raise ValueError('paddle_op must be List or string, but got type {}.'. - format(type(paddle_op))) - - if not isinstance(mapper_obj, six.class_types): - raise ValueError('mapper_obj must be Class, but got type {}.'.format( - type(mapper_obj))) - - valid_register_func = 0 - for k, v in inspect.getmembers(mapper_obj, inspect.ismethod): - if k.startswith("opset_"): - version = int(k.replace("opset_", "")) - if version > 13 or version < 1: - raise Exception( - 'the specific method of operator mapper must be named opset_[number](1<=number<=13), such as opset_9, but got {}.'. - format(k)) - valid_register_func += 1 - - if valid_register_func == 0: - raise Exception( - 'the specific method of operator mapper must be classmethod, which named opset_[number](1<=number<=13), such as opset_9, but none achieved.' - ) - - mapper = OpMapper(paddle_op_list) - mapper(mapper_obj) - - -class OpMapper(object): - OPSETS = {} - REGISTER_CUSTOM_PADDLE_OP = {} - - def __init__(self, paddle_op, **kwargs): - if not isinstance(paddle_op, list): - paddle_op = [paddle_op] - self.paddle_op = paddle_op - self.kwargs = kwargs - - def __call__(self, cls): - for k, v in inspect.getmembers(cls, inspect.ismethod): - if k.startswith("opset_"): - version = int(k.replace("opset_", "")) - for op in self.paddle_op: - if op not in OpMapper.OPSETS: - OpMapper.OPSETS[op] = {} - opset_dict = OpMapper.OPSETS[op] - opset_dict[version] = (v, self.kwargs) - - @staticmethod - def mapping(graph, node, operator_export_type="ONNX"): - try: - if node.type in OpMapper.REGISTER_CUSTOM_PADDLE_OP: - if operator_export_type in ["PaddleFallback"]: - opsets = OpMapper.OPSETS[node.type] - versions = list(opsets.keys()) - convert_version = get_max_support_version( - versions, graph.opset_version) - mapper_func, kw = opsets[convert_version] - mapper_func(graph, node, **kw) - else: - custom_paddle_op = OpMapper.REGISTER_CUSTOM_PADDLE_OP[ - node.type](node) - custom_paddle_graph, output_results = custom_paddle_op.get_paddle_graph( - ) - OpMapper.check_support_status(custom_paddle_graph.node_map, - graph.opset_version) - graph.build_op_nodes(custom_paddle_graph.node_map) - - node_output_results = dict() - for k in node.output_names: - custom_outs = output_results[k] - node_outs = node.output(k) - assert len(custom_outs) == len( - node_outs - ), "Length of custom implementation operator's outputs is not same with the length of original operator's outputs." - for i in range(len(custom_outs)): - graph.make_node( - "Identity", - inputs=[custom_outs[i]], - outputs=[node_outs[i]]) - else: - opsets = OpMapper.OPSETS[node.type] - versions = list(opsets.keys()) - convert_version = get_max_support_version(versions, - graph.opset_version) - mapper_func, kw = opsets[convert_version] - mapper_func(graph, node, **kw) - except Exception as e: - raise Exception( - "Error happened when mapping node ['{}'] to onnx, which op_type is '{}' with inputs: {} and outputs: {}, specific error: ". - format(node.layer_name, node.type, node.inputs, - node.outputs) + str(e)) - - @staticmethod - def get_recommend_opset_version(node_map, opset_version): - recommend_opset_version = OpMapper.check_support_status( - node_map, opset_version, True) - for name, node in list(node_map.items()): - if node.type in OpMapper.REGISTER_CUSTOM_PADDLE_OP: #如果是custom的op,获取custom的推荐op - custom_paddle_op = OpMapper.REGISTER_CUSTOM_PADDLE_OP[ - node.type](node) - custom_paddle_graph, output_results = custom_paddle_op.get_paddle_graph( - ) - custom_recommend_opset_version = OpMapper.check_support_status( - custom_paddle_graph.node_map, opset_version, True) - recommend_opset_version = max(recommend_opset_version, - custom_recommend_opset_version) - if opset_version != recommend_opset_version: - warning_info = "\n======================\n" - warning_info += "\nFor a successful conversion, set the recommended opset version : {}\n".format( - recommend_opset_version) - warning_info += "\n======================\n" - logging.warning(warning_info) - return recommend_opset_version - - @staticmethod - def check_support_status(node_map, opset_version, for_check=False): - op_mapping_status = { - OP_MAPPING_NO_REGISTER: [], - OP_MAPPING_NO_VERSION: [], - } - for name, node in list(node_map.items()): - if node.type in OpMapper.REGISTER_CUSTOM_PADDLE_OP: - continue - if node.type not in OpMapper.OPSETS: - op_mapping_status[OP_MAPPING_NO_REGISTER].append(node) - else: - opsets = OpMapper.OPSETS[node.type] - versions = list(opsets.keys()) - convert_version = get_max_support_version(versions, - opset_version) - if convert_version == -1: - op_mapping_status[OP_MAPPING_NO_VERSION].append(node) - - if len(op_mapping_status[OP_MAPPING_NO_REGISTER]) > 0: - unsupported_op_types = set([ - node.type for node in op_mapping_status[OP_MAPPING_NO_REGISTER] - ]) - error_info = "\nThere's {} ops are not supported yet\n".format( - len(unsupported_op_types)) - for op_type in unsupported_op_types: - error_info += "=========== {} ===========\n".format(op_type) - raise NotImplementedError(error_info) - - if len(op_mapping_status[OP_MAPPING_NO_VERSION]) > 0: - unsupported_op_types = set([ - node.type for node in op_mapping_status[OP_MAPPING_NO_VERSION] - ]) - - recommend_opset_version = -1 - for op_type in unsupported_op_types: - opsets = OpMapper.OPSETS[op_type] - if min(opsets.keys()) > recommend_opset_version: - recommend_opset_version = min(opsets.keys()) - warning_info = "\nThere are {} ops that are not supported in opset version {}, please set opset version >= {}.\n".format( - len(unsupported_op_types), opset_version, - recommend_opset_version) - - for op_type in unsupported_op_types: - warning_info += "=========== {} ===========\n".format(op_type) - if for_check: - logging.warning(warning_info) - return recommend_opset_version - raise NotImplementedError(warning_info) - return opset_version - - -class CustomPaddleOp(object): - CREATE_TIMES = {} - - def __init__(self, node): - self.main_program = paddle.static.Program() - self.startup_program = paddle.static.Program() - self.inputs = self.create_place_holder(node) - self.node = node - - def generate_scope_name(self, node): - if node.type in CustomPaddleOp.CREATE_TIMES: - CustomPaddleOp.CREATE_TIMES[node.type] += 1 - else: - CustomPaddleOp.CREATE_TIMES[node.type] = 1 - scope_prefix = node.type + str(CustomPaddleOp.CREATE_TIMES[node.type] - - 1) + '_' - return scope_prefix - - def create_place_holder(self, node): - place_holders = {} - with paddle.static.program_guard(self.main_program, - self.startup_program): - for arg_name, idxs in node.inputs.items(): - place_holders[arg_name] = [] - for idx in range(len(idxs)): - shape = node.input_shape(arg_name, idx) - dtype = node.input_dtype(arg_name, idx) - name = node.input(arg_name, idx) - data = paddle.static.data( - name=name, shape=shape, dtype=dtype) - place_holders[arg_name].append(data) - return place_holders - - def input(self, name, idx=None): - if name not in self.inputs: - return None - if idx is None: - return self.inputs[name] - if len(self.inputs[name]) <= idx: - return None - return self.inputs[name][idx] - - def get_paddle_graph(self): - scope_prefix = self.generate_scope_name(self.node) - scope = paddle.static.Scope() - with paddle.static.scope_guard(scope): - with paddle.static.program_guard(self.main_program, - self.startup_program): - with paddle.utils.unique_name.guard(scope_prefix): - res = self.forward() - feed_var_names = [ - var.name for vars in self.inputs.values() - for var in vars - ] - fetch_vars = [var for vars in res.values() for var in vars] - inference_program = graph_helper.get_program( - self.main_program, feed_var_names, fetch_vars) - paddle_graph = PaddleGraph.build_from_program( - inference_program, - feed_var_names, - fetch_vars, - scope=scope) - - output_results = dict() - for arg_name, outs in res.items(): - output_results[arg_name] = [out.name for out in outs] - return paddle_graph, output_results - - -def register_custom_paddle_op(paddle_op, custom_op): - paddle_op_list = [] - - if isinstance(paddle_op, six.string_types): - paddle_op_list.append(paddle_op) - elif isinstance(paddle_op, list): - paddle_op_list = paddle_op - else: - raise ValueError("paddle_op' must be List or string, but got type {}.". - format(type(paddle_op))) - - if not isinstance(custom_op, six.class_types): - raise ValueError("'custom_op' must be Class, but got type {}.".format( - type(custom_op))) - - forward = getattr(custom_op, "forward", None) - if not callable(forward): - raise Exception( - "Custom paddle operators must be implemented in function named 'forward'." - ) - - for op in paddle_op_list: - if op not in OpMapper.REGISTER_CUSTOM_PADDLE_OP: - OpMapper.REGISTER_CUSTOM_PADDLE_OP[op] = custom_op diff --git a/paddle2onnx/legacy/op_mapper/search.py b/paddle2onnx/legacy/op_mapper/search.py deleted file mode 100755 index 38848b5ebbb..00000000000 --- a/paddle2onnx/legacy/op_mapper/search.py +++ /dev/null @@ -1,233 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import numpy as np -from paddle2onnx.legacy.constant import dtypes -from paddle2onnx.legacy.op_mapper import OpMapper as op_mapper - - -@op_mapper('where_index') -class WhereIndex(): - support_opset_version_range = (9, 15) - - @classmethod - def opset_9(cls, graph, node, **kw): - nonzero_node = graph.make_node( - 'NonZero', inputs=node.input('Condition')) - graph.make_node( - 'Transpose', - inputs=[nonzero_node], - outputs=node.output('Out'), - perm=[1, 0]) - - -@op_mapper('top_k_v2') -class TopKV2(): - support_opset_version_range = (11, 15) - - @classmethod - def opset_11(cls, graph, node, **kw): - sorted = node.attr('sorted') - # for paddle, In gpu device, it always return the sorted value - # if not sorted: - # sorted = True - if 'K' in node.inputs and len(node.input('K')) > 0: - k_node = node.input('K', 0) - k_node_dtype = node.input_dtype('K', 0) - if dtypes.DTYPE_PADDLE_STR_MAP[k_node_dtype] != 'int64': - k_node = graph.make_node( - 'Cast', inputs=[k_node], to=dtypes.ONNX.INT64) - graph.make_node( - 'TopK', - inputs=[node.input('X', 0), k_node], - outputs=[node.output('Out', 0), node.output('Indices', 0)], - largest=node.attr('largest'), - sorted=sorted, - axis=node.attr('axis')) - else: - k = node.attr('k') - k_node = graph.make_node( - 'Constant', attrs={'dtype': dtypes.ONNX.INT64, - 'value': [k]}) - graph.make_node( - 'TopK', - inputs=[node.input('X', 0), k_node], - outputs=[node.output('Out', 0), node.output('Indices', 0)], - largest=node.attr('largest'), - sorted=sorted, - axis=node.attr('axis')) - - -@op_mapper('top_k') -class TopK(): - support_opset_version_range = (11, 15) - - @classmethod - def opset_11(cls, graph, node, **kw): - if 'K' in node.inputs and len(node.input('K')) > 0: - k_node = node.input('K', 0) - k_node_dtype = node.input_dtype('K', 0) - if dtypes.DTYPE_PADDLE_STR_MAP[k_node_dtype] != 'int64': - k_node = graph.make_node( - 'Cast', inputs=[k_node], to=dtypes.ONNX.INT64) - graph.make_node( - 'TopK', - inputs=[node.input('X', 0), k_node], - outputs=[node.output('Out', 0), node.output('Indices', 0)]) - else: - k = node.attr('k') - k_node = graph.make_node( - 'Constant', attrs={'dtype': dtypes.ONNX.INT64, - 'value': [k]}) - graph.make_node( - 'TopK', - inputs=[node.input('X', 0), k_node], - outputs=[node.output('Out', 0), node.output('Indices', 0)]) - - -@op_mapper('argsort') -class ArgSort(): - support_opset_version_range = (6, 15) - - @classmethod - def opset_10(cls, graph, node, **kw): - shape = graph.make_node('Shape', inputs=node.input('X', 0)) - from paddle2onnx.legacy.op_mapper import mapper_helper - axis = node.attr('axis') - if axis < 0: - axis = axis + len(node.input_shape('X', 0)) - dim_size = mapper_helper.slice_helper( - graph, shape, axes=[0], starts=[axis], ends=[axis + 1]) - if graph.opset_version > 10: - if not node.attr('descending'): - graph.make_node( - 'TopK', - inputs=[node.input('X', 0), dim_size], - outputs=[node.output('Out', 0), node.output('Indices', 0)], - axis=node.attr('axis'), - largest=0) - else: - graph.make_node( - 'TopK', - inputs=[node.input('X', 0), dim_size], - outputs=[node.output('Out', 0), node.output('Indices', 0)], - axis=node.attr('axis'), - largest=1) - else: - if not node.attr('descending'): - raise Exception( - "descending=False only support opset version>=11.") - else: - graph.make_node( - 'TopK', - inputs=[node.input('X', 0), dim_size], - outputs=[node.output('Out', 0), node.output('Indices', 0)], - axis=node.attr('axis')) - - @classmethod - def opset_6(cls, graph, node, **kw): - shape = node.input_shape('X', 0) - k = shape[node.attr('axis')] - assert k > 0, "while input shape is dynamic, it only support opset version>=10." - input_dtype = node.input_dtype('X', 0) - dtype = dtypes.DTYPE_PADDLE_STR_MAP[input_dtype] - inputs = node.input('X', 0) - if dtype in ["int32", "int64"]: - inputs = graph.make_node( - 'Cast', inputs=inputs, to=dtypes.ONNX.FLOAT) - if not node.attr('descending'): - raise Exception("descending=False only support opset version>=11.") - else: - output_node = node.output('Out', 0) - graph.make_node( - 'TopK', - inputs=[inputs], - outputs=[output_node, node.output('Indices', 0)], - axis=node.attr('axis'), - k=k) - if dtype in ["int32", "int64"]: - graph.make_node( - 'Cast', - inputs=[output_node], - to=dtypes.DTYPE_PADDLE_ONNX_MAP[input_dtype], - outputs=[output_node]) - - -@op_mapper('index_select') -class IndexSelect(): - support_opset_version_range = (1, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - graph.make_node( - 'Gather', - inputs=[node.input('X', 0), node.input('Index', 0)], - axis=node.attr('dim'), - outputs=node.output('Out')) - - -@op_mapper('unique') -class Unique(): - support_opset_version_range = (11, 15) - - @classmethod - def opset_11(cls, graph, node, **kw): - if node.attr('axis') == []: - graph.make_node( - 'Unique', - inputs=node.input('X'), - outputs=[ - node.output('Out', 0), node.output('Indices', 0), - node.output('Index', 0), node.output('Counts', 0) - ]) - else: - graph.make_node( - 'Unique', - inputs=node.input('X'), - axis=node.attr('axis')[0], - outputs=[ - node.output('Out', 0), node.output('Indices', 0), - node.output('Index', 0), node.output('Counts', 0) - ]) - - -@op_mapper('where') -class Where(): - support_opset_version_range = (9, 15) - - @classmethod - def opset_9(cls, graph, node, **kw): - graph.make_node( - 'Where', - inputs=[ - node.input('Condition', 0), node.input('X', 0), - node.input('Y', 0) - ], - outputs=node.output('Out')) - - -@op_mapper('masked_select') -class MaskSelect(): - support_opset_version_range = (11, 15) - - @classmethod - def opset_11(cls, graph, node, **kw): - index = graph.make_node('NonZero', inputs=node.input('Mask', 0)) - index = graph.make_node('Transpose', inputs=[index], perm=[1, 0]) - graph.make_node( - 'GatherND', - inputs=[node.input('X', 0), index], - outputs=node.output('Y')) diff --git a/paddle2onnx/legacy/op_mapper/sequence/__init__.py b/paddle2onnx/legacy/op_mapper/sequence/__init__.py deleted file mode 100644 index 847ddc47ac8..00000000000 --- a/paddle2onnx/legacy/op_mapper/sequence/__init__.py +++ /dev/null @@ -1,13 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. diff --git a/paddle2onnx/legacy/op_mapper/sequence/im2sequence.py b/paddle2onnx/legacy/op_mapper/sequence/im2sequence.py deleted file mode 100644 index accd6f13d3b..00000000000 --- a/paddle2onnx/legacy/op_mapper/sequence/im2sequence.py +++ /dev/null @@ -1,78 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License" -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import numpy as np -from paddle2onnx.legacy.constant import dtypes -from paddle2onnx.utils import logging -from paddle2onnx.legacy.op_mapper import OpMapper as op_mapper - - -@op_mapper('im2sequence') -class Im2Sequence(): - support_opset_verison_range = (1, 12) - - @classmethod - def opset_1(cls, graph, node, **kw): - n, c, h, w = node.input_shape('X', 0) - assert h > 0 and w > 0, "Only supported fixed input shape for im2sequence operator." - stride_h, stride_w = node.attr('strides') - paddings = node.attr('paddings') - assert node.attr( - 'out_stride' - ) != 1, "Only out_stride==1 is supported for im2sequence operator." - h = h + paddings[0] + paddings[1] - w = w + paddings[1] + paddings[2] - kernel_h, kernel_w = node.attr('kernels') - out_h = 1 + (h - kernel_h + stride_h - 1) // stride_h - out_w = 1 + (w - kernel_w + stride_w - 1) // stride_w - h_steps = list() - for i in range(out_h): - h_steps.append([i * stride_h, i * stride_h + kernel_h]) - w_steps = list() - for i in range(out_w): - w_steps.append([i * stride_w, i * stride_w + kernel_w]) - - slice_node_blocks = list() - for i in range(out_h): - for j in range(out_w): - starts_node = graph.make_node( - 'Constant', - dtype=dtypes.ONNX.INT64, - dims=[4], - value=[0, 0, h_steps[i][0], w_steps[j][0]]) - ends_node = graph.make_node( - 'Constant', - dtype=dtypes.ONNX.INT64, - dims=[4], - value=[999999, 999999, h_steps[i][1], w_steps[j][1]]) - nodes.extend([starts_node, ends_node]) - - slice_block_node = graph.make_node( - 'Slice', - inputs=[node.input('X', 0), starts_node, ends_node]) - flatten_block_node = graph.make_node( - "Flatten", inputs=[slice_block_node], axis=0) - nodes.extend([slice_block_node, flatten_block_node]) - concat_block_node = graph.make_node( - "Concat", - inputs=slice_node_blocks, - outputs=node.output('Out'), - axis=0) - logging.info("==========Importance Notice===========") - logging.info( - "Since im2sequence operator is used in your paddlepaddle model, the translated onnx model only support input data with batch_size=1." - ) - logging.info("======================================") diff --git a/paddle2onnx/legacy/op_mapper/tensor.py b/paddle2onnx/legacy/op_mapper/tensor.py deleted file mode 100755 index 081cbbfd216..00000000000 --- a/paddle2onnx/legacy/op_mapper/tensor.py +++ /dev/null @@ -1,2155 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import numpy as np -from paddle2onnx.legacy.constant import dtypes -from paddle2onnx.legacy.op_mapper import OpMapper as op_mapper -from paddle2onnx.legacy.op_mapper import mapper_helper -import copy -import six -import paddle - - -@op_mapper('set_value') -class SetValue(): - support_opset_version_range = (11, 15) - - @classmethod - def opset_11(cls, graph, node, **kw): - axes = node.attr('axes') - steps, is_steps_tensor = mapper_helper.get_node_attr_value( - graph, - node, - 'steps', - 'StepsTensor', - 'StepsTensorList', - return_list=True, - dtype=dtypes.ONNX.INT64) - - starts, is_starts_tensor = mapper_helper.get_node_attr_value( - graph, - node, - 'starts', - 'StartsTensor', - 'StartsTensorList', - return_list=True, - dtype=dtypes.ONNX.INT64) - - ends, is_ends_tensor = mapper_helper.get_node_attr_value( - graph, - node, - 'ends', - 'EndsTensor', - 'EndsTensorList', - return_list=True, - dtype=dtypes.ONNX.INT64) - - contain_step_bigger_than_1 = False - for i in steps: - contain_step_bigger_than_1 = i > 1 - if not isinstance(i, int) or contain_step_bigger_than_1: - contain_step_bigger_than_1 = True - break - condition = is_steps_tensor or is_starts_tensor or is_ends_tensor or contain_step_bigger_than_1 - assert not condition, "Currently not supported convert now" - - input_x_shape = node.input_shape('Input', 0) - onnx_paddings = [0] * len(input_x_shape) * 2 - value_shape = list(copy.copy(node.input_shape('Input', 0))) - for i in range(len(axes)): - axis = axes[i] - if starts[i] < 0: - starts[i] = starts[i] + input_x_shape[i] - if ends[i] < 0: - ends[i] = ends[i] + input_x_shape[i] - onnx_paddings[axis] = starts[i] - value_shape[axis] = value_shape[axis] - onnx_paddings[axis] - onnx_paddings[axis + len(input_x_shape)] = input_x_shape[ - axis] - ends[i] - if onnx_paddings[axis + len(input_x_shape)] < 0: - onnx_paddings[axis + len(input_x_shape)] = 0 - value_shape[axis] = value_shape[axis] - onnx_paddings[axis + len( - input_x_shape)] - dtype_paddle = node.input_dtype('Input', 0) - dtype = dtypes.DTYPE_PADDLE_ONNX_MAP[dtype_paddle] - value_tensor = None - shape = node.attr('shape') - if len(shape) > 0: - dtypes_list = [ - 'fp32_values', 'fp64_values', 'int32_values', 'int64_values', - 'bool_values' - ] - for i in range(len(dtypes_list)): - value = node.attr(dtypes_list[i]) - if value is not None: - break - if len(value) == 1: - total_nums = 1 - for i in value_shape: - total_nums *= i - value = value * total_nums - value_tensor = mapper_helper.constant_helper( - graph, dtype_paddle, value, shape=value_shape) - else: - value_tensor = mapper_helper.constant_helper( - graph, dtype_paddle, value, shape=shape) - else: - value_tensor = node.input('ValueTensor', 0) - MAX_FLOAT32 = 3.402823466E+38 - max_node = graph.make_node( - 'Constant', attrs={'dtype': dtype, - 'value': [MAX_FLOAT32]}) - pads_node = graph.make_node( - 'Constant', - attrs={'dtype': dtypes.ONNX.INT64, - 'value': onnx_paddings}) - value_pad_node = graph.make_node( - 'Pad', inputs=[value_tensor, pads_node, max_node]) - - condition_dtype = graph.make_node( - "Equal", inputs=[value_pad_node, max_node]) - condition_node = graph.make_node( - 'Cast', inputs=[condition_dtype], to=dtypes.ONNX.BOOL) - graph.make_node( - "Where", - inputs=[condition_node, node.input('Input', 0), value_pad_node], - outputs=node.output('Out')) - - -@op_mapper('one_hot_v2') -class OneHotV2(): - support_opset_version_range = (9, ) - - @classmethod - def opset_9(cls, graph, node, **kw): - allow_out_of_range = node.attr('allow_out_of_range') - assert not allow_out_of_range, "allow_out_of_range can not be true in one_hot_v2." - in_dtype_paddle = node.input_dtype('X', 0) - in_dtype = dtypes.DTYPE_PADDLE_ONNX_MAP[in_dtype_paddle] - out_dtype = node.output_dtype('Out', 0) - out_dtype = dtypes.DTYPE_PADDLE_ONNX_MAP[out_dtype] - inputs = node.input('X', 0) - if in_dtype_paddle == paddle.int32: - inputs = graph.make_node( - 'Cast', inputs=[inputs], to=dtypes.ONNX.INT64) - in_dtype = dtypes.ONNX.INT64 - value_node = graph.make_node('Constant', dtype=out_dtype, value=[0, 1]) - depth = node.attr('depth') - if node.input('depth_tensor', 0) is not None: - depth_node = node.input('depth_tensor', 0) - else: - depth_node = graph.make_node( - 'Constant', dtype=in_dtype, value=[depth]) - reshaped_input_node = graph.make_node( - 'OneHot', - inputs=[inputs, depth_node, value_node], - outputs=node.output('Out')) - - -@op_mapper('concat') -class Concat(): - support_opset_version_range = (4, 15) - - @classmethod - def opset_4(cls, graph, node, **kw): - inputs = node.input('X') - - input_dtypes = [node.input_dtype('X', i) for i in range(len(inputs))] - inputs = mapper_helper.dtype_alignment(graph, inputs, input_dtypes) - node_axis = node.input('AxisTensor') - if node_axis is not None and len(node_axis) > 0: - axis_node = node.input('AxisTensor')[0] - try: - axis = mapper_helper.get_value_from_parameters(graph, - axis_node)[0] - except Exception as e: - raise Exception( - "Currently does not support the axis parameter as input tensor" - + str(e)) - else: - axis = node.attr('axis') - if axis < 0: - axis = axis + len(node.input_shape('X', 0)) - - node = graph.make_node( - 'Concat', inputs=inputs, outputs=node.output('Out'), axis=axis) - - -@op_mapper('assign') -class Assign(): - support_opset_version_range = (1, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - inputs = node.input('X') - graph.make_node('Identity', inputs=inputs, outputs=node.output('Out')) - - -@op_mapper('lod_reset') -class LodReset(): - support_opset_version_range = (1, ) - - @classmethod - def opset_1(cls, graph, node, **kw): - graph.make_node( - 'Identity', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('eye') -class Eye(): - support_opset_version_range = (9, ) - - @classmethod - def opset_9(cls, graph, node, **kw): - num_rows = node.attr('num_rows') - num_columns = node.attr('num_columns') - dtype = node.output_dtype('Out', 0) - value = [0] * num_rows * num_columns - value_tensor = mapper_helper.constant_helper( - graph, dtype, value, shape=[num_rows, num_columns]) - graph.make_node( - 'EyeLike', inputs=[value_tensor], outputs=node.output('Out')) - - -@op_mapper('stack') -class Stack(): - support_opset_version_range = (4, 15) - - @classmethod - def opset_4(cls, graph, node, **kw): - inputs = node.input('X') - input_dtypes = [node.input_dtype('X', i) for i in range(len(inputs))] - inputs = mapper_helper.dtype_alignment(graph, inputs, input_dtypes) - axis = node.attr('axis') - - unsqueezed_inputs = list() - for ipt in inputs: - unsqueezed_ipt = mapper_helper.unsqueeze_helper(graph, ipt, [axis]) - unsqueezed_inputs.append(unsqueezed_ipt) - graph.make_node( - 'Concat', - inputs=unsqueezed_inputs, - outputs=node.output('Y'), - axis=axis) - - -@op_mapper('unstack') -class Unstack(): - support_opset_version_range = (2, 15) - - @classmethod - def opset_2(cls, graph, node, **kw): - axis = node.attr('axis') - ndim = node.block.vars[node.input('X')[0]].ndim - axis = axis + ndim if axis < 0 else axis - output_y = mapper_helper.split_helper( - graph, - node.input('X'), - axis=axis, - split=[1] * len(node.output('Y')), - outputs=len(node.output('Y'))) - - if isinstance(output_y, six.string_types): - output_y = [output_y] - - for i in range(len(output_y)): - mapper_helper.squeeze_helper(graph, output_y[i], [axis], - node.output('Y', i)) - - -@op_mapper('expand_as_v2') -class ExpandAsV2(): - support_opset_version_range = (8, 15) - - @classmethod - def opset_8(cls, graph, node, **kw): - target_shape = node.attr('target_shape') - if node.input('target_tensor', 0) is not None: - target_shape = graph.make_node( - 'Shape', inputs=[node.input('target_tensor', 0)]) - elif target_shape is not None: - target_shape = graph.make_node( - 'Constant', - attrs={'dtype': dtypes.ONNX.INT64, - 'value': target_shape}) - else: - raise Exception( - "Not find attribute: 'target_shape' or tensor 'target_tensor'") - node = graph.make_node( - 'Expand', - inputs=[node.input('X', 0), target_shape], - outputs=node.output('Out')) - - -@op_mapper('expand_v2') -class ExpandV2(): - support_opset_version_range = (8, 15) - - @classmethod - def opset_8(cls, graph, node, **kw): - expand_shape, _ = mapper_helper.get_node_attr_value( - graph, - node, - 'shape', - 'Shape', - 'expand_shapes_tensor', - dtype=dtypes.ONNX.INT64) - - input_shape = node.input_shape('X', 0) - input_shape_node = graph.make_node('Shape', inputs=node.input('X', 0)) - - node_shape = node.attr('shape') - node_shape_tensor = node.input('Shape') - node_shape_tensor_list = node.input('expand_shapes_tensor') - if node_shape_tensor is not None and len(node_shape_tensor) > 0: - diff = node.input_shape('Shape', 0)[0] - len(input_shape) - elif node_shape_tensor_list is not None and \ - len(node_shape_tensor_list) > 0: - diff = len(node_shape_tensor_list) - len(input_shape) - elif node_shape is not None and len(node_shape) > 0: - diff = len(node_shape) - len(input_shape) - expand_shape = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=expand_shape) - - if diff > 0: - one_node = graph.make_node( - 'Constant', - attrs={'dtype': dtypes.ONNX.INT64, - 'value': [1] * diff}) - input_shape_node = graph.make_node( - 'Concat', inputs=[one_node, input_shape_node], axis=0) - - if graph.opset_version < 12: - input_shape_node = graph.make_node( - 'Cast', inputs=[input_shape_node], to=dtypes.ONNX.FLOAT) - expand_shape = graph.make_node( - 'Cast', inputs=[expand_shape], to=dtypes.ONNX.FLOAT) - shape = graph.make_node( - 'Max', inputs=[input_shape_node, expand_shape]) - shape = graph.make_node( - 'Cast', inputs=[shape], to=dtypes.ONNX.INT64) - else: - shape = graph.make_node( - 'Max', inputs=[input_shape_node, expand_shape]) - node = graph.make_node( - 'Expand', - inputs=[node.input('X', 0), shape], - outputs=node.output('Out')) - - -@op_mapper('shape') -class Shape(): - support_opset_version_range = (6, 15) - - @classmethod - def opset_6(cls, graph, node, **kw): - shape_node = graph.make_node('Shape', inputs=node.input('Input')) - graph.make_node( - 'Cast', - inputs=[shape_node], - outputs=node.output('Out'), - to=dtypes.ONNX.INT32) - - -@op_mapper('size') -class Numel(): - supports_opset_version_range = (1, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - size_node = graph.make_node('Size', inputs=node.input('Input')) - mapper_helper.unsqueeze_helper(graph, size_node, [0], - node.output('Out')) - - -@op_mapper('split') -class Split(): - support_opset_version_range = (1, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - sections = node.attr('sections') - axis = cls.get_axis(graph, node) - if isinstance(sections, list) and len(sections) == 1: - graph.make_node( - 'Identity', inputs=node.input('X'), outputs=node.output('Out')) - else: - if len(sections) > 0: - input_shape = node.block.vars[node.input('X')[0]].shape - section_index = [ - i for i, val in enumerate(sections) if val == -1 - ] - if input_shape[axis] != -1 and len(section_index) == 1: - sections[section_index[0]] = input_shape[axis] - sum( - sections) - 1 - mapper_helper.split_helper( - graph, - node.input('X'), - axis=axis, - split=sections, - outputs=node.output('Out')) - else: - graph.make_node( - 'Split', - inputs=node.input('X'), - outputs=node.output('Out'), - axis=axis) - - @classmethod - def get_axis(cls, graph, node): - if len(node.input('AxisTensor')) > 0: - axis_node = node.input('AxisTensor')[0] - # When axis is tensor, only int32 and int64 are supported - if axis_node not in graph.parameters: - raise Exception( - "Currently does not support the axis parameter as input tensor!" - ) - else: - axis = graph.parameters[axis_node].attribute[0].t.int32_data - if axis is None or len(axis) < 1: - axis = graph.parameters[axis_node].attribute[ - 0].t.int64_data[0] - else: - axis = node.attr('axis') - return axis - - -@op_mapper(['roll']) -class Roll(): - support_opset_version_range = (4, 15) - - @classmethod - def roll(cls, graph, node, input_x, dims, shifts): - for i in range(len(dims)): - if graph.opset_version >= 10 and isinstance(shifts, - six.string_types): - to_dtype = dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype( - 'ShiftsTensor', 0)] - const_i = graph.make_node('Constant', dtype=to_dtype, value=i) - const_0 = graph.make_node('Constant', dtype=to_dtype, value=0) - shift_node = graph.make_node( - 'Gather', inputs=[shifts, const_i], axis=0) - shift_node = graph.make_node( - "Sub", inputs=[const_0, shift_node]) - shift_node = mapper_helper.unsqueeze_helper(graph, shift_node, - [0]) - elif graph.opset_version < 10 and isinstance(shifts, - six.string_types): - raise Exception( - "shifts of roll is Tensor, please try with higher onnx opset_version>=10." - ) - else: - shift_node = [-shifts[i]] - to_dtype = dtypes.ONNX.INT64 - shapes = [] - shape = mapper_helper.slice_helper( - graph, input_x, [dims[i]], shift_node, [60000], dtype=to_dtype) - shapes.append(shape) - shape = mapper_helper.slice_helper( - graph, input_x, [dims[i]], [0], shift_node, dtype=to_dtype) - shapes.append(shape) - input_x = graph.make_node('Concat', inputs=shapes, axis=dims[i]) - return input_x - - @classmethod - def flatten(cls, graph, node): - dims = len(node.input_shape('X', 0)) - start_axis = 0 - end_axis = dims - 1 - shape_node = graph.make_node('Shape', inputs=node.input('X')) - if end_axis < dims - 1: - slice1 = mapper_helper.slice_helper( - graph, shape_node, axes=[0], starts=[0], ends=[start_axis]) - slice3 = mapper_helper.slice_helper( - graph, shape_node, axes=[0], starts=[end_axis + 1], - ends=[dims]) - slices = [ - slice1, graph.make_node( - 'Constant', value=[-1], dtype=dtypes.ONNX.INT64), slice3 - ] - else: - slice1 = mapper_helper.slice_helper( - graph, shape_node, axes=[0], starts=[0], ends=[start_axis]) - slices = [ - slice1, graph.make_node( - 'Constant', value=[-1], dtype=dtypes.ONNX.INT64) - ] - final_shape = graph.make_node('Concat', inputs=slices, axis=0) - output = graph.make_node( - 'Reshape', inputs=[node.input('X')[0], final_shape]) - return output - - @classmethod - def opset_4(cls, graph, node, **kw): - dims = node.attr('axis') - shifts = node.attr('shifts') - input_x = node.input('X')[0] - input_shape = node.input_shape('X', 0) - shifts_node = node.input('ShiftsTensor') - if len(dims) > 0: - axes = [ - axis + len(input_shape) if axis < 0 else axis - for i, axis in enumerate(dims) - ] - if shifts_node is not None and len(shifts_node) > 0: - shifts = shifts_node[0] - else: - for i in range(0, len(axes)): - if input_shape[axes[i]] > 0: - assert -input_shape[axes[i]] <= shifts[i] <= input_shape[axes[i]], \ - "the value of shifts in axis is less than the value of input_shape in axis." - - input_x = cls.roll(graph, node, input_x, axes, shifts) - graph.make_node( - 'Identity', inputs=[input_x], outputs=node.output('Out')) - else: - if shifts_node is not None and len(shifts_node) > 0: - shifts = shifts_node[0] - input_x = cls.flatten(graph, node) - input_x = cls.roll(graph, node, input_x, [0], shifts) - shape_node = graph.make_node( - 'Constant', - attrs={'dtype': dtypes.ONNX.INT64, - 'value': list(input_shape)}) - graph.make_node( - 'Reshape', - inputs=[input_x, shape_node], - outputs=node.output('Out')) - - -@op_mapper(['slice', 'strided_slice']) -class Slice(): - support_opset_version_range = (1, 15) - - @classmethod - def decrease_axis(cls, node): - # tensor[i,:] will decrease rank of origin input, example: - # paddle.slice() will not decrease rank of origin input - # if input shape is [2, 3], input[0, :] will generate output with shape [3], not [1, 3]. - # paddle.slice(input, 0, 1, 0) will generate output with shape [1, 3], not [3]. - - decrease_axis = node.attr('decrease_axis') - if len(decrease_axis) == 0: - return None - if node.output_shape('Out', 0) == [0]: - return decrease_axis - if len(node.input_shape('Input', 0)) > len(node.output_shape('Out', 0)): - return decrease_axis - return None - - @classmethod - def opset_1(cls, graph, node, **kw): - axes = node.attr('axes') - strides, strides_is_tensor = mapper_helper.get_node_attr_value( - graph, node, 'strides', 'StridesTensor', 'StridesTensorList', True) - strides = [1] * len(axes) if strides is None else strides - steps = [i for i, val in enumerate(strides) if val == 1] - assert len(steps) == len(axes), \ - "Slice in onnx(opset<10) not support attribute 'step', Try converting with opset_version >=10" - - starts, start_is_tensor = mapper_helper.get_node_attr_value( - graph, node, 'starts', 'StartsTensor', 'StartsTensorList', True) - ends, end_is_tensor = mapper_helper.get_node_attr_value( - graph, node, 'ends', 'EndsTensor', 'EndsTensorList', True) - - assert not strides_is_tensor and not start_is_tensor and not end_is_tensor, \ - "Slice in onnx(opset<10) not support attribute 'steps','starts' or 'ends' which have tensor value, " \ - "Try converting with opset_version >=10 " - - decrease_axis = cls.decrease_axis(node) - if decrease_axis is None: - graph.make_node( - "Slice", - inputs=[node.input('Input')[0]], - outputs=node.output('Out'), - axes=axes, - starts=starts, - ends=ends) - else: - sliced = graph.make_node( - "Slice", - inputs=[node.input('Input')[0]], - axes=axes, - starts=starts, - ends=ends) - mapper_helper.squeeze_helper(graph, sliced, decrease_axis, - node.output('Out')) - - @classmethod - def opset_10(cls, graph, node, **kw): - axes = node.attr('axes') - strides, _ = mapper_helper.get_node_attr_value( - graph, - node, - 'strides', - 'StridesTensor', - 'StridesTensorList', - dtype=dtypes.ONNX.INT64) - strides = [1] * len(axes) if strides is None else strides - - starts, _ = mapper_helper.get_node_attr_value( - graph, - node, - 'starts', - 'StartsTensor', - 'StartsTensorList', - dtype=dtypes.ONNX.INT64) - ends, _ = mapper_helper.get_node_attr_value( - graph, - node, - 'ends', - 'EndsTensor', - 'EndsTensorList', - dtype=dtypes.ONNX.INT64) - - if isinstance(starts, list): - starts_node = graph.make_node( - 'Constant', - attrs={'dtype': dtypes.ONNX.INT64, - 'value': starts}) - else: - starts_node = starts - if isinstance(ends, list): - ends_node = graph.make_node( - 'Constant', attrs={'dtype': dtypes.ONNX.INT64, - 'value': ends}) - else: - ends_node = ends - - if isinstance(strides, list): - strides_node = graph.make_node( - 'Constant', - attrs={'dtype': dtypes.ONNX.INT64, - 'value': strides}) - else: - strides_node = strides - - steps_node = strides_node - axes_node = graph.make_node( - 'Constant', attrs={'dtype': dtypes.ONNX.INT64, - 'value': axes}) - - decrease_axis = cls.decrease_axis(node) - if decrease_axis is None: - sliced = graph.make_node( - "Slice", - inputs=[ - node.input('Input')[0], starts_node, ends_node, axes_node, - steps_node - ], - outputs=node.output('Out')) - else: - sliced = graph.make_node( - "Slice", - inputs=[ - node.input('Input')[0], starts_node, ends_node, axes_node, - steps_node - ]) - mapper_helper.squeeze_helper(graph, sliced, decrease_axis, - node.output('Out')) - - -@op_mapper(['sequence_expand']) -class SequenceExpand(): - support_opset_version_range = () - - @classmethod - def opset_1(cls, graph, node, **kw): - graph.make_node( - 'Identity', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper(['expand']) -class Expand(): - support_opset_version_range = (6, 15) - - @classmethod - def opset_6(cls, graph, node, **kw): - expand_times, _ = mapper_helper.get_node_attr_value( - graph, - node, - 'expand_times', - 'ExpandTimes', - 'expand_times_tensor', - dtype=dtypes.ONNX.INT64) - - if isinstance(expand_times, list): - expand_times = graph.make_node( - 'Constant', - attrs={'dtype': dtypes.ONNX.INT64, - 'value': expand_times}) - - graph.make_node( - "Tile", - inputs=[node.input('X', 0), expand_times], - outputs=node.output('Out')) - - -@op_mapper(['tile']) -class Tile(): - support_opset_version_range = (6, 15) - - @classmethod - def opset_6(cls, graph, node, **kw): - repeat_times, _ = mapper_helper.get_node_attr_value( - graph, - node, - 'repeat_times', - 'RepeatTimes', - 'repeat_times_tensor', - dtype=dtypes.ONNX.INT64) - - if isinstance(repeat_times, list): - repeat_times = graph.make_node( - 'Constant', - attrs={'dtype': dtypes.ONNX.INT64, - 'value': repeat_times}) - - graph.make_node( - "Tile", - inputs=[node.input('X', 0), repeat_times], - outputs=node.output('Out')) - - -@op_mapper('range') -class Range(): - support_opset_version_range = (11, 15) - - @classmethod - def opset_11(cls, graph, node, **kw): - start = node.input('Start', 0) - end = node.input('End', 0) - step = node.input('Step', 0) - start_t = mapper_helper.squeeze_helper(graph, start, [0]) - end_t = mapper_helper.squeeze_helper(graph, end, [0]) - step_t = mapper_helper.squeeze_helper(graph, step, [0]) - graph.make_node( - "Range", - inputs=[start_t, end_t, step_t], - outputs=node.output('Out')) - - -@op_mapper('fill_constant') -class Constant(): - support_opset_version_range = (1, 15) - - @classmethod - def check_int_type(cls, dtype): - if dtype in [dtypes.ONNX.INT16, dtypes.ONNX.INT32, dtypes.ONNX.INT64]: - return True - return False - - @classmethod - def opset_1(cls, graph, node, **kw): - value = node.attr('value') - dtype = node.attr('dtype') - value_is_scalar_tensor = False - if 'ValueTensor' in node.inputs and len(node.input('ValueTensor')) > 0: - rank = len(node.input_shape("ValueTensor", 0)) - if rank == 1 and node.input_shape("ValueTensor", 0)[0] == 1: - value_is_scalar_tensor = True - value = node.input("ValueTensor")[0] - else: - raise Exception( - "paddle.full with tensor value parameter is not supported yet." - ) - - shape, is_shape_tensor = mapper_helper.get_node_attr_value( - graph, - node, - 'shape', - 'ShapeTensor', - 'ShapeTensorList', - dtype=dtypes.ONNX.INT64) - - if graph.opset_version >= 9 and (is_shape_tensor or - value_is_scalar_tensor): - if not is_shape_tensor: - shape = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=shape) - input_dtype = dtypes.DTYPE_PADDLE_ONNX_MAP[dtype] - if not value_is_scalar_tensor and cls.check_int_type(input_dtype): - to_dtype = dtypes.ONNX.DOUBLE - outputs = None - else: - to_dtype = input_dtype - outputs = node.output('Out') - - if value_is_scalar_tensor: - base_value = graph.make_node( - 'ConstantOfShape', - inputs=shape, - attrs={'dims': [1], - 'dtype': to_dtype, - 'value': 0}) - node2 = graph.make_node( - "Add", inputs=[base_value, value], outputs=outputs) - else: - node2 = graph.make_node( - 'ConstantOfShape', - inputs=shape, - outputs=outputs, - attrs={'dims': [1], - 'dtype': to_dtype, - 'value': value}) - - if not value_is_scalar_tensor and cls.check_int_type(input_dtype): - graph.make_node( - 'Cast', - inputs=node2, - outputs=node.output('Out'), - attrs={'to': input_dtype}) - else: - assert not is_shape_tensor and not value_is_scalar_tensor, \ - "Currently op ['fill_constant'] does not support in onnx(opset<9) when 'shape' or 'fill_value' has " \ - "tensor, Try converting with opset_version >=9 " - - value = np.ones(shape) * value - value = value.astype(dtypes.DTYPE_PADDLE_NUMPY_MAP[dtype]) - value = value.flatten().tolist() - - graph.make_node( - 'Constant', - inputs=[], - outputs=node.output('Out'), - attrs={ - 'dims': shape, - 'dtype': dtypes.DTYPE_PADDLE_ONNX_MAP[dtype], - 'value': value - }) - - -@op_mapper(['lookup_table_v2', 'lookup_table']) -class Embedding(): - support_opset_version_range = (1, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - ids = node.input('Ids', 0) - if node.type == 'lookup_table' and node.input_shape('Ids', 0)[-1] == 1: - ids = mapper_helper.squeeze_helper(graph, - node.input('Ids', 0), [-1]) - padding_idx = node.attr('padding_idx') - input_shape = node.input_shape('W', 0) - if padding_idx != -1: - key = node.input('W', 0) - if -1 in input_shape: - assert False, "opset version < 11 do not support padding_idx !=-1 and weight is tensor with dynamic shape, please set opset version > 11 or use input_spec to set input shape" - else: - data = np.ones(shape=input_shape, dtype=np.float32) - data[padding_idx] = 0.0 - dtype = dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('W', 0)] - constant = graph.make_node( - 'Constant', - dtype=dtype, - dims=input_shape, - value=data.flatten().tolist()) - weight_node = graph.make_node( - 'Mul', inputs=[node.input('W', 0), constant]) - graph.make_node( - 'Gather', - inputs=[weight_node, ids], - outputs=node.output('Out')) - else: - graph.make_node( - 'Gather', - inputs=[node.input('W', 0), ids], - outputs=node.output('Out')) - - @classmethod - def opset_11(cls, graph, node, **kw): - ids = node.input('Ids', 0) - if node.type == 'lookup_table' and node.input_shape('Ids', 0)[-1] == 1: - ids = mapper_helper.squeeze_helper(graph, - node.input('Ids', 0), [-1]) - - padding_idx = node.attr('padding_idx') - input_shape = node.input_shape('W', 0) - if padding_idx != -1: - if -1 in input_shape: - replace_shape = list(copy.copy(input_shape)) - del (replace_shape[0]) - replace_data = graph.make_node( - 'Constant', - dtype=dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('W', - 0)], - dims=replace_shape, - value=[0.0] * np.prod(replace_shape)) - index = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[padding_idx]) - Scatter_node = graph.make_node( - 'ScatterND', - inputs=[node.input('W', 0), index, replace_data]) - graph.make_node( - 'Gather', - inputs=[Scatter_node, ids], - outputs=node.output('Out')) - else: - data = np.ones(shape=input_shape, dtype=np.float32) - data[padding_idx] = 0.0 - dtype = dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('W', 0)] - constant = graph.make_node( - 'Constant', - dtype=dtype, - dims=input_shape, - value=data.flatten().tolist()) - weight_node = graph.make_node( - 'Mul', inputs=[node.input('W', 0), constant]) - graph.make_node( - 'Gather', - inputs=[weight_node, ids], - outputs=node.output('Out')) - else: - graph.make_node( - 'Gather', - inputs=[node.input('W', 0), ids], - outputs=node.output('Out')) - - -@op_mapper('fill_constant_batch_size_like') -class FillConstantBatchSizeLike(): - support_opset_version_range = (9, 12) - - @classmethod - def opset_10(cls, graph, node, **kw): - out_shape = node.attr('shape') - input_dim_idx = node.attr('input_dim_idx') - output_dim_idx = node.attr('output_dim_idx') - - del out_shape[output_dim_idx] - out_shape.insert(0, 1) - - dtype = dtypes.DTYPE_PADDLE_ONNX_MAP[node.attr('dtype')] - if node.attr("str_value") is not None and node.attr("str_value") != "": - value = eval(node.attr("str_value")) - else: - value = node.attr('value') - input_shape = node.input_shape('Input', 0) - constant = graph.make_node( - 'Constant', - dtype=dtype, - dims=out_shape, - value=[value] * np.prod(out_shape)) - - shape = graph.make_node('Shape', inputs=node.input('Input')) - start = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[input_dim_idx]) - end = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT64, value=[input_dim_idx + 1]) - batch = graph.make_node('Slice', inputs=[shape, start, end]) - repeat = batch - if len(out_shape) > 1: - repeat = graph.make_node( - 'Constant', - dtype=dtypes.ONNX.INT64, - value=[1] * (len(out_shape) - 1)) - repeat = graph.make_node('Concat', inputs=[batch, repeat], axis=-1) - if output_dim_idx == 0: - graph.make_node( - 'Tile', inputs=[constant, repeat], outputs=node.output('Out')) - else: - out = graph.make_node('Tile', inputs=[constant, repeat]) - perm = list(range(len(out_shape))) - del perm[0] - perm.insert(output_dim_idx, 0) - graph.make_node( - 'Transpose', - inputs=[out], - perm=perm, - outputs=node.output('Out')) - - -@op_mapper('fill_any_like') -class FullLike(): - ''' - fill_any_like is kernel for paddle op::full_like & ones_like - ''' - support_opset_version_range = (9, 15) - - @classmethod - def opset_9(cls, graph, node, **kw): - shape_node = graph.make_node('Shape', inputs=node.input('X')) - value = node.attr('value') - dtype = node.attr('dtype') - input_dtype = node.input_dtype('X', 0) - if dtype is None: - dtype = input_dtype - np_dtype = dtypes.DTYPE_PADDLE_STR_MAP[dtype] - onnx_dtype = dtypes.DTYPE_PADDLE_ONNX_MAP[dtype] - graph.make_node( - 'ConstantOfShape', - inputs=[shape_node], - outputs=node.output('Out'), - dims=[1], - dtype=onnx_dtype, - value=np.array(value).astype(np_dtype).tolist()) - - -@op_mapper('fill_zeros_like') -class FullZeroLike(): - ''' - fill_zeros_like is kernel for paddle op::zeros_like - ''' - support_opset_version_range = (9, 15) - - @classmethod - def opset_9(cls, graph, node, **kw): - shape_node = graph.make_node('Shape', inputs=node.input('X')) - value = 0 - dtype = node.attr('dtype') - input_dtype = node.input_dtype('X', 0) - if dtype is None: - dtype = input_dtype - np_dtype = dtypes.DTYPE_PADDLE_STR_MAP[dtype] - onnx_dtype = dtypes.DTYPE_PADDLE_ONNX_MAP[dtype] - graph.make_node( - 'ConstantOfShape', - inputs=[shape_node], - outputs=node.output('Out'), - dims=[1], - dtype=onnx_dtype, - value=np.array(value).astype(np_dtype).tolist()) - - -@op_mapper('gather_nd') -class Gather_nd(): - support_opset_version_range = (11, 15) - - @classmethod - def opset_11(cls, graph, node, **kw): - data = node.input('X', 0) - index = node.input('Index', 0) - index_dtype = node.input_dtype('Index', 0) - index_node = None - if index_dtype != paddle.int64: - index_node = graph.make_node( - 'Cast', inputs=[node.input('Index', 0)], to=dtypes.ONNX.INT64) - else: - index_node = index - graph.make_node( - 'GatherND', inputs=[data, index_node], outputs=node.output('Out')) - - -@op_mapper('gather') -class Gather(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - axis = node.attr('axis') - if node.input('Axis', 0) != None: - axis_node = node.input('Axis', 0) - try: - axis = mapper_helper.get_value_from_parameters(graph, - axis_node)[0] - except Exception as e: - raise Exception( - "Currently does not support the axis parameter as input tensor" - + str(e)) - if axis is None: - axis = 0 - if len(node.input_shape('Index', 0)) == 1: - # gather - graph.make_node( - 'Gather', - inputs=[node.input('X', 0), node.input('Index', 0)], - outputs=node.output('Out'), - attrs={'axis': axis}) - else: - raise Exception( - "please try to convert OP:gather(indices's rank >1) with opset_version >= 11." - ) - - @classmethod - def opset_11(cls, graph, node, **kw): - axis = node.attr('axis') - if node.input('Axis', 0) != None: - axis_node = node.input('Axis', 0) - try: - axis = mapper_helper.get_value_from_parameters(graph, - axis_node)[0] - except Exception as e: - raise Exception( - "Currently does not support the axis parameter as input tensor" - + str(e)) - if axis is None: - axis = 0 - if len(node.input_shape('Index', 0)) == 1: - # gather - graph.make_node( - 'Gather', - inputs=[node.input('X', 0), node.input('Index', 0)], - outputs=node.output('Out'), - attrs={'axis': axis}) - else: - # gather_nd - index_dtype = node.input_dtype('Index', 0) - if index_dtype != paddle.int64: - index_node = graph.make_node( - 'Cast', - inputs=[node.input('Index', 0)], - to=dtypes.ONNX.INT64) - graph.make_node( - 'GatherND', - inputs=[node.input('X', 0), index_node], - outputs=node.output('Out')) - else: - graph.make_node( - 'GatherND', - inputs=[node.input('X', 0), node.input('Index', 0)], - outputs=node.output('Out')) - - -@op_mapper('squeeze2') -class Squeeze(): - support_opset_version_range = (1, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - shape = node.input_shape('X', 0) - ret = [i for i, val in enumerate(shape) if val > 1] - if len(ret) == len(shape): - graph.make_node( - 'Identity', inputs=node.input('X'), outputs=node.output('Out')) - else: - axes = cls.compute_axes(graph, node) - if len(axes) > 0: - axes.sort() - mapper_helper.squeeze_helper(graph, - node.input('X', 0), axes, - node.output('Out')) - else: - graph.make_node( - 'Squeeze', - inputs=[node.input('X', 0)], - outputs=node.output('Out')) - - @classmethod - def compute_axes(cls, graph, node): - shape = node.input_shape('X', 0) - axes = node.attr('axes') - if len(axes) > 0: - axes = [ - axis + len(shape) if axis < 0 else axis - for i, axis in enumerate(axes) - ] - return axes - - -@op_mapper('assign_value') -class Assign(): - support_opset_version_range = (1, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - if len(node.input_names) > 0: - graph.make_node( - 'Identity', inputs=node.input('X'), outputs=node.output('Out')) - else: - parameters = {} - value = np.array(node.attr('fp32_values')) - if value is None or value.size < 1: - value = np.array(node.attr('int32_values')) - if value is None or value.size < 1: - value = np.array(node.attr('int64_values')) - parameter = { - 'data': value, - 'dtype': node.output_dtype("Out", 0), - 'shape': node.attr('shape') - } - parameters[node.output('Out', 0)] = parameter - graph.build_parameters(parameters) - - -@op_mapper('transpose2') -class Transpose(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - graph.make_node( - 'Transpose', - inputs=node.input('X'), - outputs=node.output('Out'), - perm=node.attr('axis')) - - -@op_mapper('flatten2') -class Flatten(): - support_opset_version_range = (1, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - input_dtype = dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)] - if input_dtype in [dtypes.ONNX.INT32, dtypes.ONNX.INT64 - ] and graph.opset_version < 9: - raise Exception( - "int32 or int64 not supported in onnx <9, please try with higher onnx opset_version>=9." - ) - - graph.make_node( - 'Flatten', - inputs=node.input('X'), - outputs=node.output('Out'), - axis=node.attr('axis')) - - -@op_mapper('flatten_contiguous_range') -class FlattenContiguousRange(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - dims = len(node.input_shape('X', 0)) - start_axis = node.attr('start_axis') - end_axis = node.attr('stop_axis') - shape_node = graph.make_node('Shape', inputs=node.input('X')) - if start_axis < 0: - start_axis += dims - if end_axis < 0: - end_axis += dims - if start_axis == 0 and end_axis == dims - 1: - final_shape = graph.make_node( - 'Constant', value=[-1], dtype=dtypes.ONNX.INT64) - elif start_axis == 0: - slice_end = mapper_helper.slice_helper( - graph, shape_node, axes=[0], starts=[end_axis + 1], - ends=[dims]) - slices = [ - graph.make_node( - 'Constant', value=[-1], dtype=dtypes.ONNX.INT64), slice_end - ] - final_shape = graph.make_node('Concat', inputs=slices, axis=0) - elif end_axis == dims - 1: - slice_start = mapper_helper.slice_helper( - graph, shape_node, axes=[0], starts=[0], ends=[start_axis]) - slices = [ - slice_start, graph.make_node( - 'Constant', value=[-1], dtype=dtypes.ONNX.INT64) - ] - final_shape = graph.make_node('Concat', inputs=slices, axis=0) - else: - slice_start = mapper_helper.slice_helper( - graph, shape_node, axes=[0], starts=[0], ends=[start_axis]) - slice_end = mapper_helper.slice_helper( - graph, shape_node, axes=[0], starts=[end_axis + 1], - ends=[dims]) - slices = [ - slice_start, graph.make_node( - 'Constant', value=[-1], dtype=dtypes.ONNX.INT64), slice_end - ] - final_shape = graph.make_node('Concat', inputs=slices, axis=0) - graph.make_node( - 'Reshape', - inputs=[node.input('X')[0], final_shape], - outputs=node.output('Out')) - - -@op_mapper('reshape2') -class Reshape(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - shape_name = 'ShapeTensor' - if shape_name not in node.inputs or len(node.input(shape_name)) == 0: - shape_name = 'Shape' - if shape_name not in node.inputs or len(node.input(shape_name)) == 0: - if node.attr('shape') is None or len(node.attr('shape')) == 0: - raise Exception("shape tensor and shape attrubite all unkown.") - if len(node.input(shape_name)) > 1: - dims = [] - for i in range(len(node.input(shape_name))): - dim = node.input(shape_name)[i] - dim = graph.make_node( - 'Cast', inputs=[dim], to=dtypes.ONNX.INT64) - dims.append(dim) - shape = graph.make_node('Concat', inputs=dims, axis=-1) - graph.make_node( - 'Reshape', - inputs=[node.input('X')[0], shape], - outputs=node.output('Out')) - elif len(node.input(shape_name)) == 1: - cast_shape_node = graph.make_node( - 'Cast', inputs=node.input(shape_name), to=dtypes.ONNX.INT64) - graph.make_node( - 'Reshape', - inputs=[node.input('X')[0], cast_shape_node], - outputs=node.output('Out')) - elif node.attr('shape') is not None and len(node.attr('shape')) > 0: - shape_node = graph.make_node( - 'Constant', - attrs={ - 'dtype': dtypes.ONNX.INT64, - 'value': node.attr('shape') - }) - reshape_node = graph.make_node( - 'Reshape', - inputs=[node.input('X')[0], shape_node], - outputs=node.output('Out')) - - -@op_mapper('unsqueeze2') -class Unsqueeze(): - support_opset_version_range = (1, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - axes = cls.get_axes(graph, node) - mapper_helper.unsqueeze_helper(graph, - node.input('X'), axes, - node.output('Out')) - - @classmethod - def opset_13(cls, graph, node, **kw): - axes_node = cls.get_axes(graph, node, return_node=True) - graph.make_node( - 'Unsqueeze', - inputs=node.input('X') + [axes_node], - outputs=node.output('Out')) - - @classmethod - def get_axes(cls, graph, node, return_node=False): - axes_node = None - ndim = node.block.vars[node.input('X')[0]].ndim - if len(node.attr('axes')) > 0: - axes = node.attr('axes') - else: - axes_node = node.input('AxesTensor')[0] - if axes_node is not None and graph.opset_version > 12 and return_node: - return axes_node - try: - axes = mapper_helper.get_value_from_parameters(graph, axes_node) - except Exception as e: - raise Exception( - "Currently does not support the axes parameter as input tensor in onnx(opset<13), " - "Try converting with opset_version >=13 " + str(e)) - # axes is list of non-negative integers - axes = [ - axis + ndim + i + 1 if axis < 0 else axis - for i, axis in enumerate(axes) - ] - - axes_copy = axes.copy() - assert sorted( - axes) == axes_copy, "axes must be arranged in the following order" - assert len(set(axes)) == len(axes), "axes have duplicate axis" - - if return_node: - if axes_node is None: - axes_node = graph.make_node( - 'Constant', - attrs={'dtype': dtypes.ONNX.INT64, - 'value': axes}) - return axes_node - return axes - - -@op_mapper('reciprocal') -class Reciprocal(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - graph.make_node( - 'Reciprocal', inputs=node.input('X'), outputs=node.output('Out')) - - -@op_mapper('cast') -class Cast(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - graph.make_node( - 'Cast', - inputs=node.input('X'), - outputs=node.output('Out'), - to=dtypes.DTYPE_PADDLE_ONNX_MAP[node.attr('out_dtype')]) - - -@op_mapper('linspace') -class Linspace(): - support_opset_version_range = (9, 15) - - @classmethod - def opset_9(cls, graph, node, **kw): - start = node.input('Start', 0) - stop = node.input('Stop', 0) - num = node.input('Num', 0) - dtype = node.attr('dtype') - - start = graph.make_node('Cast', inputs=[start], to=dtypes.ONNX.FLOAT) - stop = graph.make_node('Cast', inputs=[stop], to=dtypes.ONNX.FLOAT) - - sub_a_node = graph.make_node('Sub', inputs=[stop, start]) - - one_node = graph.make_node( - 'Constant', - dtype=dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('Num', 0)], - value=[1]) - - sub_b_node = graph.make_node('Sub', inputs=[num, one_node]) - - sub_b_float_node = graph.make_node( - 'Cast', inputs=[sub_b_node], to=dtypes.ONNX.FLOAT) - - step = graph.make_node('Div', inputs=[sub_a_node, sub_b_float_node]) - - range_tensor = graph.make_node( - 'Cast', inputs=[num], to=dtypes.ONNX.INT64) - - one_like_node = graph.make_node( - 'ConstantOfShape', - inputs=[range_tensor], - dtype=dtypes.ONNX.FLOAT, - value=[1]) - - none_zero_node = graph.make_node('NonZero', inputs=[one_like_node]) - - trans_none_zero_node = graph.make_node( - 'Transpose', inputs=[none_zero_node], perm=[1, 0]) - - trans_squeeze = mapper_helper.squeeze_helper(graph, - trans_none_zero_node, [1]) - - trans_squeeze = graph.make_node( - 'Cast', inputs=[trans_squeeze], to=dtypes.ONNX.FLOAT) - - mul_node = graph.make_node('Mul', inputs=[trans_squeeze, step]) - - add_node = graph.make_node('Add', inputs=[mul_node, start]) - graph.make_node( - 'Cast', - inputs=[add_node], - outputs=node.output('Out'), - to=dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('Start', 0)]) - - -@op_mapper('clip') -class Clip(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - min_value = node.attr('min') - max_value = node.attr('max') - if node.input('Max', 0) is None or len(node.input('Max')) == 0: - max_ = max_value - else: - max_ = node.input('Max', 0) - if node.input('Min', 0) is None or len(node.input('Min')) == 0: - min_ = min_value - else: - min_ = node.input('Min', 0) - mapper_helper.clip_helper(graph, node, - node.input('X', 0), max_, min_, - node.output('Out', 0)) - - -@op_mapper(['pad2d', 'pad3d']) -class Pad(): - support_opset_version_range = (1, 12) - - @classmethod - def opset_1(cls, graph, node, **kw): - if node.attr('mode') == 'replicate': - mode = 'edge' - elif node.attr('mode') == 'circular': - raise Exception("The padding mode = circular is not supported, " \ - "Please try the other three ways") - else: - mode = node.attr('mode') - pads = cls.convert_padding(node, **kw) - if pads is None: - key = node.input('Paddings', 0) - padding = None - if key in graph.parameters.keys(): - paddings = graph.parameters[key].attribute[0].t.int32_data - if node.attr('data_format') == 'NCHW': - pads = [ - 0, 0, paddings[0], paddings[2], 0, 0, paddings[1], - paddings[3] - ] - elif node.attr('data_format') == 'NHWC': - pads = [ - 0, paddings[0], paddings[2], 0, 0, paddings[1], - paddings[3], 0 - ] - elif node.attr('data_format') == 'NCDHW': - pads = [ - 0, 0, paddings[4], paddings[2], paddings[0], 0, 0, - paddings[5], paddings[3], paddings[1] - ] - elif node.attr('data_format') == 'NDHWC': - pads = [ - 0, paddings[4], paddings[2], paddings[0], 0, 0, - paddings[5], paddings[3], paddings[1], 0 - ] - else: - raise Exception("In Pad op, padding can not be tensor" \ - "Please set opset version >= 11") - - value = None - if node.attr('pad_value') is not None: - value = node.attr('pad_value') - elif node.attr('value') is not None: - value = node.attr('value') - graph.make_node( - 'Pad', - inputs=node.input('X'), - outputs=node.output('Out'), - mode=mode, - value=value, - pads=pads) - - @classmethod - def opset_11(cls, graph, node, **kw): - pads = cls.convert_padding(node, **kw) - if node.attr('mode') == 'replicate': - mode = 'edge' - elif node.attr('mode') == 'circular': - raise Exception("The padding mode = circular is not supported, " \ - "Please try the other three ways") - else: - mode = node.attr('mode') - pads_node = None - if isinstance(pads, list): - pads_node = graph.make_node( - 'Constant', attrs={'dtype': dtypes.ONNX.INT64, - 'value': pads}) - else: - key = node.input('Paddings', 0) - padding = None - if key in graph.parameters.keys(): - paddings = graph.parameters[key].attribute[0].t.int32_data - onnx_paddings = None - if node.attr('data_format') == 'NCHW': - onnx_paddings = [ - 0, 0, paddings[0], paddings[2], 0, 0, paddings[1], - paddings[3] - ] - elif node.attr('data_format') == 'NHWC': - onnx_paddings = [ - 0, paddings[0], paddings[2], 0, 0, paddings[1], - paddings[3], 0 - ] - elif node.attr('data_format') == 'NCDHW': - onnx_paddings = [ - 0, 0, paddings[4], paddings[2], paddings[0], 0, 0, - paddings[5], paddings[3], paddings[1] - ] - elif node.attr('data_format') == 'NDHWC': - onnx_paddings = [ - 0, paddings[4], paddings[2], paddings[0], 0, 0, - paddings[5], paddings[3], paddings[1], 0 - ] - - pads_node = graph.make_node( - 'Constant', - attrs={'dtype': dtypes.ONNX.INT64, - 'value': onnx_paddings}) - else: - padding_node = node.input('Paddings', 0) - casted_padding_node = graph.make_node( - 'Cast', inputs=[padding_node], to=dtypes.ONNX.FLOAT) - zero_node = None - if node.attr('data_format') == 'NCHW' or node.attr( - 'data_format') == 'NHWC': - zero_node = graph.make_node( - 'Constant', dtype=dtypes.ONNX.FLOAT, value=[0] * 8) - else: - zero_node = graph.make_node( - 'Constant', dtype=dtypes.ONNX.FLOAT, value=[0] * 10) - index = None - if node.attr('data_format') == 'NCHW': - index = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT32, - value=[2, 6, 3, 7]) - elif node.attr('data_format') == 'NHWC': - index = graph.make_node( - 'Constant', dtype=dtypes.ONNX.INT32, - value=[1, 5, 2, 6]) - elif node.attr('data_format') == 'NCDHW': - index = graph.make_node( - 'Constant', - dtype=dtypes.ONNX.INT32, - value=[4, 9, 3, 8, 2, 7]) - elif node.attr('data_format') == 'NDHWC': - index = graph.make_node( - 'Constant', - dtype=dtypes.ONNX.INT32, - value=[3, 8, 2, 7, 1, 6]) - - float_paddle_node = graph.make_node( - 'ScatterElements', - inputs=[zero_node, index, casted_padding_node]) - paddle_node = graph.make_node( - 'Cast', inputs=[float_paddle_node], to=dtypes.ONNX.INT64) - pads_node = paddle_node - - value = None - if node.attr('pad_value') is not None: - value = node.attr('pad_value') - elif node.attr('value') is not None: - value = node.attr('value') - value_node = graph.make_node( - 'Constant', - attrs={ - 'dtype': dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('X', 0)], - 'value': value - }) - - graph.make_node( - 'Pad', - inputs=node.input('X') + [pads_node, value_node], - outputs=node.output('Out'), - mode=mode) - - @classmethod - def convert_padding(cls, node, **kw): - x_shape = node.input_shape('X', 0) - paddings = node.attr('paddings') - if paddings == []: - return None - onnx_paddings = None - if node.attr('data_format') == 'NCHW': - onnx_paddings = [ - 0, 0, paddings[0], paddings[2], 0, 0, paddings[1], paddings[3] - ] - elif node.attr('data_format') == 'NHWC': - onnx_paddings = [ - 0, paddings[0], paddings[2], 0, 0, paddings[1], paddings[3], 0 - ] - elif node.attr('data_format') == 'NCDHW': - onnx_paddings = [ - 0, 0, paddings[4], paddings[2], paddings[0], 0, 0, paddings[5], - paddings[3], paddings[1] - ] - elif node.attr('data_format') == 'NDHWC': - onnx_paddings = [ - 0, paddings[4], paddings[2], paddings[0], 0, 0, paddings[5], - paddings[3], paddings[1], 0 - ] - return onnx_paddings - - -@op_mapper('gaussian_random') -class GaussianRandom(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - shape_input_list = node.input('ShapeTensorList') - shape_input = None - if len(shape_input_list) == 0: - shape_input = node.input('ShapeTensor') - else: - shape_input = graph.make_node( - "Concat", inputs=node.input('ShapeTensorList'), axis=0) - if shape_input is None or len(shape_input) == 0: - graph.make_node( - 'RandomNormal', - dtype=dtypes.DTYPE_PADDLE_ONNX_MAP[node.attr('dtype')], - outputs=node.output('Out'), - shape=node.attr('shape'), - seed=float(node.attr('seed')), - mean=node.attr('mean'), - scale=node.attr('std')) - else: - cast_input_shape = graph.make_node( - 'Cast', inputs=shape_input, to=dtypes.ONNX.INT64) - zero_like_node = graph.make_node( - 'ConstantOfShape', - inputs=cast_input_shape, - dims=[1], - dtype=dtypes.ONNX.FLOAT, - value=[0]) - graph.make_node( - 'RandomNormalLike', - dtype=dtypes.DTYPE_PADDLE_ONNX_MAP[node.attr('dtype')], - outputs=node.output('Out'), - inputs=zero_like_node, - seed=float(node.attr('seed')), - mean=node.attr('mean'), - scale=node.attr('std')) - - -@op_mapper('uniform_random_batch_size_like') -class UniformRandom(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_1(cls, graph, node, **kw): - graph.make_node( - 'RandomUniformLike', - inputs=node.input('Input'), - outputs=node.output('Out'), - high=node.attr('max'), - dtype=dtypes.DTYPE_PADDLE_ONNX_MAP[node.attr('dtype')], - low=node.attr('min'), - seed=float(node.attr('seed')), ) - - -@op_mapper('uniform_random') -class UniformRandom(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - shape_input_list = node.input('ShapeTensorList') - shape_input = None - if len(shape_input_list) == 0: - shape_input = node.input('ShapeTensor') - else: - shape_input = graph.make_node( - "Concat", inputs=node.input('ShapeTensorList'), axis=0) - if shape_input is None or len(shape_input) == 0: - graph.make_node( - 'RandomUniform', - dtype=dtypes.DTYPE_PADDLE_ONNX_MAP[node.attr('dtype')], - outputs=node.output('Out'), - shape=node.attr('shape'), - seed=float(node.attr('seed')), - low=node.attr('min'), - high=node.attr('max')) - else: - cast_input_shape = graph.make_node( - 'Cast', inputs=shape_input, to=dtypes.ONNX.INT64) - zero_like_node = graph.make_node( - 'ConstantOfShape', - inputs=cast_input_shape, - dtype=dtypes.ONNX.FLOAT, - value=[0]) - graph.make_node( - 'RandomUniformLike', - dtype=dtypes.DTYPE_PADDLE_ONNX_MAP[node.attr('dtype')], - outputs=node.output('Out'), - inputs=zero_like_node, - seed=float(node.attr('seed')), - low=node.attr('min'), - high=node.attr('max')) - - -# 'bilinear_interp', 'nearest_interp', scale only support 2, 4, 6, 8, 10 -@op_mapper( - [ - 'bilinear_interp', 'nearest_interp', 'bilinear_interp_v2', - 'nearest_interp_v2', 'bicubic_interp_v2', 'linear_interp_v2', - 'trilinear_interp_v2', 'trilinear_interp', 'linear_interp' - ], - mapper_dict={ - 'bilinear_interp': 'linear', - 'nearest_interp': 'nearest', - 'bilinear_interp_v2': 'linear', - 'nearest_interp_v2': 'nearest', - 'bicubic_interp_v2': 'cubic', - 'linear_interp_v2': 'linear', - 'trilinear_interp_v2': 'linear', - 'trilinear_interp': 'linear', - 'linear_interp': 'linear', - }, - opset_op_dict={ - 9: 'Upsample', - 10: 'Resize', - }) -class Resize(): - support_opset_version_range = (9, 15) - - @classmethod - def opset_9(cls, graph, node, **kw): - inputs = [node.input('X')[0]] - resize_type = kw['mapper_dict'][node.type] - cls.waringInfo(graph, node, resize_type) - if len(node.input('OutSize')) > 0 or len(node.input('SizeTensor')) > 0: - output_node = cls.compute_outsize_node( - graph, node, return_scale=True) - elif 'Scale' in node.inputs and len(node.input('Scale')) > 0: - output_node = cls.compute_scale_node(graph, node) - else: - output_node = cls.compute_attrs_node(graph, node, return_scale=True) - - inputs = inputs + output_node - op = kw['opset_op_dict'][graph.opset_version] - graph.make_node( - op, inputs=inputs, outputs=node.output('Out'), mode=resize_type) - - @classmethod - def opset_11(cls, graph, node, **kw): - inputs = [node.input('X')[0]] - resize_type = kw['mapper_dict'][node.type] - cls.waringInfo(graph, node, resize_type) - if node.attr('align_corners'): - coordinate_transformation_mode = 'align_corners' - elif node.attr('align_mode') == 1 and resize_type is not 'cubic': - coordinate_transformation_mode = 'asymmetric' - elif resize_type == 'nearest': - coordinate_transformation_mode = 'asymmetric' - else: - coordinate_transformation_mode = 'half_pixel' - roi_node = graph.make_node( - 'Constant', - attrs={ - 'dtype': dtypes.ONNX.FLOAT, - 'value': [1, 1, 1, 1, 1, 1, 1, 1] - }) - - inputs.append(roi_node) - if len(node.input('OutSize')) > 0 or len(node.input('SizeTensor')) > 0: - output_node = cls.compute_outsize_node(graph, node) - elif 'Scale' in node.inputs and len(node.input('Scale')) > 0: - output_node = cls.compute_scale_node(graph, node) - else: - output_node = cls.compute_attrs_node(graph, node) - inputs = inputs + output_node - attrs = { - 'mode': resize_type, - 'coordinate_transformation_mode': coordinate_transformation_mode - } - if resize_type == 'nearest' and coordinate_transformation_mode == 'asymmetric': - attrs['nearest_mode'] = 'floor' - graph.make_node( - 'Resize', inputs=inputs, outputs=node.output('Out'), attrs=attrs) - - @classmethod - def compute_outsize_node(cls, graph, node, return_scale=False): - dtype = dtypes.ONNX.INT64 - if return_scale: - dtype = dtypes.ONNX.FLOAT - input_shape_node = graph.make_node('Shape', inputs=node.input('X')) - if dtype != dtypes.ONNX.INT64: - input_shape_node = graph.make_node( - 'Cast', inputs=[input_shape_node], to=dtype) - shape_pre_node = mapper_helper.slice_helper( - graph, input_shape_node, axes=[], starts=[0], ends=[2]) - - out_size = [node.attr('out_d'), node.attr('out_h'), node.attr('out_w')] - out_size = [val for val in out_size if val > 0] - use_tensor = False - if len(node.input('OutSize')) > 0 or len(node.input('SizeTensor')) > 0: - use_tensor = True - if len(out_size) > 0 and not use_tensor: - out_size_node = graph.make_node( - 'Constant', attrs={'dtype': dtype, - 'value': out_size}) - else: - out_size_node, _ = mapper_helper.get_node_attr_value( - graph, node, None, 'OutSize', 'SizeTensor', dtype=dtype) - out_size_node = graph.make_node( - 'Concat', inputs=[shape_pre_node, out_size_node], axis=0) - - if return_scale: - scale_node = graph.make_node( - 'Div', inputs=[out_size_node, input_shape_node]) - return [scale_node] - - scale_empty_node = graph.make_node( - 'Constant', attrs={'dtype': dtypes.ONNX.FLOAT, - 'value': []}) - return [scale_empty_node, out_size_node] - - @classmethod - def compute_scale_node(cls, graph, node): - cast_scale = graph.make_node( - 'Cast', inputs=node.input('Scale'), to=dtypes.ONNX.FLOAT) - inputs_cocat = [] - const_node = graph.make_node( - 'Constant', attrs={'dtype': dtypes.ONNX.FLOAT, - 'value': [1, 1]}) - inputs_cocat.append(const_node) - scale = node.attr('scale') - if isinstance(scale, (float, int)): - cast_scale = [cast_scale] * (len(node.input_shape('X', 0)) - 2) - inputs_cocat = inputs_cocat + cast_scale - else: - inputs_cocat = inputs_cocat + [cast_scale] - scale_node = graph.make_node('Concat', inputs=inputs_cocat, axis=0) - return [scale_node] - - @classmethod - def compute_attrs_node(cls, graph, node, return_scale=False): - out_size = [node.attr('out_d'), node.attr('out_h'), node.attr('out_w')] - scale = node.attr('scale') - if isinstance(scale, (float, int)): - scale = [scale] * (len(node.input_shape('X', 0)) - 2) - - out_size = [val for val in out_size if val > 0] - if len(out_size) > 0: - output_node = cls.compute_outsize_node( - graph, node, return_scale=return_scale) - return output_node - - assert len(scale) > 0, Exception("scale size should > 0!") - scale_node = graph.make_node( - 'Constant', - attrs={'dtype': dtypes.ONNX.FLOAT, - 'value': [1, 1] + scale}) - return [scale_node] - - @classmethod - def waringInfo(cls, graph, node, resize_type): - assert node.attrs['data_layout'] == 'NCHW', \ - "The conv data layout should be 'NCHW' , but received data format " \ - "is %s." % node.attrs['data_format'] - - if graph.opset_version < 11: - if node.attr('align_corners') or resize_type in ["cubic"]: - raise Exception( - "When align_corners is true or resize_type is 'cubic', the case isn't supported in onnx(opset<=10), " - "Try converting with opset_version>= 11 ") - if node.attr('align_mode') == 0 and resize_type in [ - "bilinear", "linear", "trilinear" - ]: - raise Exception( - "When align_mode == 0 and resize_type is 'bilinear' or 'linear or 'trilinear', the case isn't " - "supported in onnx(opset<=10), Try converting with opset_version>= 11 " - ) - - -@op_mapper('pixel_shuffle') -class PixelShuffle(): - support_opset_version_range = (11, 15) - - @classmethod - def opset_11(cls, graph, node, **kw): - upscale_factor = node.attr('upscale_factor') - - node = graph.make_node( - 'DepthToSpace', - inputs=node.input('X'), - outputs=node.output('Out'), - blocksize=upscale_factor, - mode='CRD') - - -@op_mapper('scatter') -class Scatter(): - support_opset_version_range = (11, 15) - - @classmethod - def opset_11(cls, graph, node, **kw): - ids = node.input('Ids', 0) - input_dtype = dtypes.DTYPE_PADDLE_ONNX_MAP[node.input_dtype('Ids', 0)] - if input_dtype != dtypes.ONNX.INT64: - ids = graph.make_node('Cast', inputs=[ids], to=dtypes.ONNX.INT64) - - shape = graph.make_node( - 'Constant', - value=[node.input_shape('Ids', 0)[0], 1], - dtype=dtypes.ONNX.INT64) - reshape_index = graph.make_node('Reshape', inputs=[ids, shape]) - if not node.attr('overwrite'): - raise Exception("overwrite = False not support yet.") - else: - graph.make_node( - 'ScatterND', - inputs=[ - node.input('X', 0), reshape_index, node.input('Updates', 0) - ], - outputs=node.output('Out')) - - -@op_mapper('scatter_nd_add') -class ScatterndAdd(): - support_opset_version_range = (11, 12) - - @classmethod - def opset_11(cls, graph, node, **kw): - shape = graph.make_node('Shape', inputs=node.input('X', 0)) - zero_like_node = graph.make_node( - 'ConstantOfShape', - inputs=[shape], - dims=[1], - dtype=dtypes.ONNX.FLOAT, - value=[0]) - add_node = graph.make_node( - 'ScatterND', - inputs=[ - zero_like_node, node.input('Index', 0), node.input('Updates', 0) - ], ) - graph.make_node( - 'Add', - inputs=[node.input('X', 0), add_node], - outputs=node.output('Out')) - - -@op_mapper('meshgrid') -class Meshgrid(): - support_opset_version_range = (8, 15) - - @classmethod - def opset_8(cls, graph, node, **kw): - tensors = [t for t in list(node.input('X'))] - tensors_shape = [graph.make_node('Shape', inputs=t) for t in tensors] - out_shape = graph.make_node('Concat', inputs=tensors_shape, axis=0) - out = [] - for i, t in enumerate(tensors): - shape_i = [ - graph.make_node( - 'Constant', - attrs={'dtype': dtypes.ONNX.INT64, - 'value': [1]}) - ] * len(tensors) - shape_i[i] = tensors_shape[i] - t_reshaped = graph.make_node( - 'Reshape', - inputs=[t, graph.make_node( - 'Concat', inputs=shape_i, axis=0)]) - out.append( - graph.make_node( - 'Expand', - inputs=[t_reshaped, out_shape], - outputs=node.output('Out')[i])) - - -@op_mapper('flip') -class Flip(): - support_opset_version_range = (7, 15) - - @classmethod - def opset_7(cls, graph, node, **kw): - inputs = node.input('X') - x_dtype = node.input_dtype('X', 0) - if x_dtype == paddle.bool or x_dtype == paddle.float64: - inputs = [ - graph.make_node( - "Cast", inputs=inputs, to=dtypes.ONNX.FLOAT) - ] - axes = node.attr("axis") - if not isinstance(axes, list): - axes = [axes] - input_shape = node.input_shape('X', 0) - - for i, axis in enumerate(axes): - if axis < 0: - axes[i] += len(input_shape) - assert input_shape[ - axis] > 0, "The dimension in axis of input must be fixed for flip operator, but now the input shape({}) in axis({}) is unknow.".format( - input_shape, axis) - - temp_input = inputs[0] - for i, axis in enumerate(axes): - if input_shape[axis] == 1: - if i != len(axes) - 1: - continue - else: - if x_dtype == paddle.bool or x_dtype == paddle.float64: - graph.make_node( - "Cast", - inputs=[temp_input], - outputs=node.output("Out"), - to=dtypes.DTYPE_PADDLE_ONNX_MAP[x_dtype]) - else: - graph.make_node( - "Identity", - inputs=[temp_input], - outputs=node.output("Out")) - else: - splits = graph.make_node( - "Split", - inputs=[temp_input], - outputs=input_shape[axis], - axis=axis, - split=[1] * input_shape[axis]) - reversed_splits = splits[::-1] - if i != len(axes) - 1: - temp_input = graph.make_node( - "Concat", inputs=reversed_splits, axis=axis) - else: - if x_dtype == paddle.bool or x_dtype == paddle.float64: - out = graph.make_node( - "Concat", inputs=reversed_splits, axis=axis) - graph.make_node( - "Cast", - inputs=[out], - outputs=node.output("Out"), - to=dtypes.DTYPE_PADDLE_ONNX_MAP[x_dtype]) - else: - graph.make_node( - "Concat", - inputs=reversed_splits, - outputs=node.output("Out"), - axis=axis) - - @classmethod - def opset_13(cls, graph, node, **kw): - inputs = node.input('X') - x_dtype = node.input_dtype('X', 0) - if x_dtype == paddle.bool or x_dtype == paddle.float64: - inputs = [ - graph.make_node( - "Cast", inputs=inputs, to=dtypes.ONNX.FLOAT) - ] - axes = node.attr("axis") - if not isinstance(axes, list): - axes = [axes] - input_shape = node.input_shape('X', 0) - - for i, axis in enumerate(axes): - if axis < 0: - axes[i] += len(input_shape) - assert input_shape[ - axis] > 0, "The dimension in axis of input must be fixed for flip operator, but now the input shape({}) in axis({}) is unknow.".format( - input_shape, axis) - - temp_input = inputs[0] - for i, axis in enumerate(axes): - if input_shape[axis] == 1: - if i != len(axes) - 1: - continue - else: - if x_dtype == paddle.bool or x_dtype == paddle.float64: - graph.make_node( - "Cast", - inputs=[temp_input], - outputs=node.output("Out"), - to=dtypes.DTYPE_PADDLE_ONNX_MAP[x_dtype]) - else: - graph.make_node( - "Identity", - inputs=[temp_input], - outputs=node.output("Out")) - else: - split = graph.make_node( - 'Constant', - attrs={ - 'dtype': dtypes.ONNX.INT64, - 'value': [1] * input_shape[axis] - }) - splits = graph.make_node( - "Split", - inputs=[temp_input, split], - outputs=input_shape[axis], - axis=axis) - reversed_splits = splits[::-1] - if i != len(axes) - 1: - temp_input = graph.make_node( - "Concat", inputs=reversed_splits, axis=axis) - else: - if x_dtype == paddle.bool or x_dtype == paddle.float64: - out = graph.make_node( - "Concat", inputs=reversed_splits, axis=axis) - graph.make_node( - "Cast", - inputs=[out], - outputs=node.output("Out"), - to=dtypes.DTYPE_PADDLE_ONNX_MAP[x_dtype]) - else: - graph.make_node( - "Concat", - inputs=reversed_splits, - outputs=node.output("Out"), - axis=axis) diff --git a/paddle2onnx/legacy/passes/__init__.py b/paddle2onnx/legacy/passes/__init__.py deleted file mode 100755 index 9ac316d8415..00000000000 --- a/paddle2onnx/legacy/passes/__init__.py +++ /dev/null @@ -1,17 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from .pass_manager import PassManager -from .inplace_node_pass import InplaceNodePass -from .dumplicate_names_pass import DumplicateNamesPass \ No newline at end of file diff --git a/paddle2onnx/legacy/passes/dumplicate_names_pass.py b/paddle2onnx/legacy/passes/dumplicate_names_pass.py deleted file mode 100755 index 82d17a129de..00000000000 --- a/paddle2onnx/legacy/passes/dumplicate_names_pass.py +++ /dev/null @@ -1,91 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from paddle2onnx.legacy.passes import PassManager -from paddle2onnx.utils import logging - - -@PassManager('dumplicate_names_pass') -class DumplicateNamesPass(object): - - name_count = dict() - - @classmethod - def generate_new_name(cls, name): - for saved_name in cls.name_count: - if name.startswith(saved_name): - cls.name_count[saved_name] += 1 - new_name = saved_name + '.' + str(cls.name_count[saved_name]) - return new_name - cls.name_count[name] = 1 - new_name = name + '.' + str(cls.name_count[name]) - return new_name - - @classmethod - def run_pass(cls, onnx_graph): - renamer = {} - tensor_names = set() - for name, node in onnx_graph.parameters.items(): - output = node.output - for opt in output: - assert opt not in tensor_names, "There's dumplicate names in parameters." - tensor_names.add(opt) - - for ipt in onnx_graph.input_nodes: - assert ipt.name not in tensor_names, "There's dumplicate names in exported parameters and inputs." - tensor_names.add(ipt.name) - - for name, node in onnx_graph.node_map.items(): - inputs = node.inputs - outputs = node.outputs - update_node = False - for idx in range(len(inputs)): - ipt = inputs[idx] - if ipt not in renamer: - continue - updated_name = renamer[ipt] - while updated_name in renamer: - updated_name = renamer[updated_name] - inputs[idx] = updated_name - update_node = True - - for idx in range(len(outputs)): - opt = outputs[idx] - if opt not in tensor_names: - tensor_names.add(opt) - continue - renamed_tensor_name = opt - while renamed_tensor_name in renamer: - renamed_tensor_name = renamer[renamed_tensor_name] - new_name = cls.generate_new_name(renamed_tensor_name) - logging.warning("[Renamer Pass] Will rename {}, to {}".format( - renamed_tensor_name, new_name)) - outputs[idx] = new_name - update_node = True - renamer[renamed_tensor_name] = new_name - - if update_node: - node.set_inputs(inputs) - node.set_outputs(outputs) - onnx_graph.update_node(node) - - for opt in onnx_graph.output_nodes: - if opt.name not in renamer: - continue - updated_name = renamer[opt.name] - while updated_name in renamer: - updated_name = renamer[updated_name] - opt.name = updated_name - - return onnx_graph diff --git a/paddle2onnx/legacy/passes/inplace_node_pass.py b/paddle2onnx/legacy/passes/inplace_node_pass.py deleted file mode 100755 index e9fc6f84bed..00000000000 --- a/paddle2onnx/legacy/passes/inplace_node_pass.py +++ /dev/null @@ -1,62 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from paddle2onnx.legacy.passes import PassManager - - -def get_repeated_output(inputs, outputs): - repeated_output = {} - for idx in range(len(outputs)): - opt = outputs[idx] - if opt in inputs: - repeated_output[opt] = idx - return repeated_output - - -@PassManager('inplace_node_pass') -class InplaceNodePass(object): - - name_count = dict() - - @classmethod - def generate_new_name(cls, name): - if name in cls.name_count: - cls.name_count[name] += 1 - else: - cls.name_count[name] = 1 - new_name = name + '.' + str(cls.name_count[name]) - return new_name - - @classmethod - def run_pass(cls, onnx_graph): - node_map = list(onnx_graph.node_map.items()) - name_mapping = {} - for idx in range(len(node_map)): - name, node = node_map[idx] - inputs = node.inputs - outputs = node.outputs - for idx in range(len(inputs)): - ipt = inputs[idx] - if ipt in name_mapping: - inputs[idx] = name_mapping[ipt] - repeated_output = get_repeated_output(inputs, outputs) - if len(repeated_output) != 0: - for opt, idx in repeated_output.items(): - name_mapping[opt] = cls.generate_new_name(opt) - outputs[idx] = name_mapping[opt] - node.set_inputs(inputs) - node.set_outputs(outputs) - onnx_graph.update_node(node) - - return onnx_graph diff --git a/paddle2onnx/legacy/passes/pass_manager.py b/paddle2onnx/legacy/passes/pass_manager.py deleted file mode 100644 index ca8813c4499..00000000000 --- a/paddle2onnx/legacy/passes/pass_manager.py +++ /dev/null @@ -1,39 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -import inspect - - -class PassManager(object): - PASSES = {} - - def __init__(self, name, **kwargs): - self.name = name - self.kwargs = kwargs - - def __call__(self, cls): - for k, v in inspect.getmembers(cls, inspect.ismethod): - if k == 'run_pass': - self.PASSES[self.name] = (v, self.kwargs) - - @staticmethod - def run_pass(graph, custom_pass_list): - for pass_name in custom_pass_list: - try: - pass_func, kw = PassManager.PASSES[pass_name] - pass_func(graph, **kw) - except: - raise Exception("Error happened when excute pass: {}".format( - pass_name)) - return graph diff --git a/paddle2onnx/mapper/activation.cc b/paddle2onnx/mapper/activation.cc deleted file mode 100644 index 10932bd3864..00000000000 --- a/paddle2onnx/mapper/activation.cc +++ /dev/null @@ -1,473 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. -#include "paddle2onnx/mapper/activation.h" - -namespace paddle2onnx { - -REGISTER_MAPPER(relu, ActivationMapper) -REGISTER_MAPPER(relu6, Relu6Mapper) -REGISTER_MAPPER(tanh, ActivationMapper) -REGISTER_MAPPER(log, ActivationMapper) -REGISTER_MAPPER(sigmoid, ActivationMapper) -REGISTER_MAPPER(sqrt, ActivationMapper) -REGISTER_MAPPER(softplus, ActivationMapper) -REGISTER_MAPPER(exp, ActivationMapper) -REGISTER_MAPPER(floor, ActivationMapper) -REGISTER_MAPPER(cos, ActivationMapper) -REGISTER_MAPPER(sin, ActivationMapper) -REGISTER_MAPPER(round, ActivationMapper) -REGISTER_MAPPER(abs, ActivationMapper) -REGISTER_MAPPER(acos, ActivationMapper) -REGISTER_MAPPER(asin, ActivationMapper) -REGISTER_MAPPER(atan, ActivationMapper) -REGISTER_MAPPER(sinh, ActivationMapper) -REGISTER_MAPPER(tan, ActivationMapper) -REGISTER_MAPPER(ceil, ActivationMapper) -REGISTER_MAPPER(cosh, ActivationMapper) -REGISTER_MAPPER(softsign, ActivationMapper) -REGISTER_MAPPER(sign, ActivationMapper) -REGISTER_MAPPER(erf, ActivationMapper) -REGISTER_MAPPER(reciprocal, ActivationMapper) -REGISTER_MAPPER(leaky_relu, LeakyReluMapper) -REGISTER_MAPPER(gelu, GeluMapper) -REGISTER_MAPPER(selu, SeluMapper) -REGISTER_MAPPER(prelu, PReluMapper) -REGISTER_MAPPER(hard_sigmoid, HardSigmoidMapper) -REGISTER_MAPPER(swish, SwishMapper) -REGISTER_MAPPER(hard_swish, HardSwishMapper) -REGISTER_MAPPER(softmax, SoftMaxMapper) -REGISTER_MAPPER(brelu, BReluMapper) -REGISTER_MAPPER(elu, EluMapper) -REGISTER_MAPPER(hard_shrink, HardShrinkMapper) -REGISTER_MAPPER(softshrink, SoftShrinkMapper) -REGISTER_MAPPER(mish, MishMapper) -REGISTER_MAPPER(square, SquareMapper) -REGISTER_MAPPER(size, SizeMapper) -REGISTER_MAPPER(rsqrt, RsqrtMapper) -REGISTER_MAPPER(logsigmoid, LogSigmoidMapper) -REGISTER_MAPPER(log_softmax, LogSoftmaxMapper) -REGISTER_MAPPER(tanh_shrink, TanhShrinkMapper) -REGISTER_MAPPER(thresholded_relu, ThresholdedReluMapper) -REGISTER_MAPPER(log1p, Log1PMapper) -REGISTER_MAPPER(log2, Log2Mapper) -REGISTER_MAPPER(log10, Log10Mapper) -REGISTER_MAPPER(silu, SiluMapper) - -int32_t ActivationMapper::GetMinOpset(bool verbose) { - if (OpType() == "softplus") { - float beta = 0.0; - float threshold = 20.0; - GetAttr("beta", &beta); - GetAttr("threshold", &threshold); - if ((beta - 1.0) > 1e-06 || (beta - 1.0) < -1e-06 || - (threshold - 20.0) > 1e-06 || (threshold - 20.0) < -1e-06) { - Error() << "Only support softplus with beta == 1.0 and threshold == 20.0." - << std::endl; - return -1; - } - } - if (OpType() == "round") { - Logger(verbose, 11) << RequireOpset(11) << std::endl; - return 11; - } - if (OpType() == "sinh" || OpType() == "cosh" || OpType() == "sign") { - Logger(verbose, 9) << RequireOpset(9) << std::endl; - return 9; - } - return 7; -} - -void ActivationMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - auto iter = op_mapper_.find(OpType()); - Assert(op_mapper_.end() != iter, - "Cannot find " + OpType() + " in activation op_mapper."); - if (OpType() == "erf") { - auto input = helper_->AutoCast(input_info[0].name, input_info[0].dtype, - P2ODataType::FP32); - auto output = helper_->MakeNode(iter->second, {input})->output(0); - helper_->AutoCast(output, output_info[0].name, P2ODataType::FP32, - output_info[0].dtype); - } else { - helper_->MakeNode(iter->second, {input_info[0].name}, - {output_info[0].name}); - } -} - -void Relu6Mapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - float min = 0.0; - helper_->Clip(input_info[0].name, output_info[0].name, min, threshold_, - input_info[0].dtype); -} - -int32_t PReluMapper::GetMinOpset(bool verbose) { - auto input_info = GetInput("X"); - auto slope_info = GetInput("Alpha"); - if (input_info[0].Rank() != slope_info[0].Rank()) { - if (slope_info[0].Rank() > 1) { - Error() - << "Only support rank of alpha <=1 while Rank(alpha) != Rank(input)." - << std::endl; - return -1; - } - } - return 7; -} - -void PReluMapper::Opset7() { - auto input_info = GetInput("X"); - auto slope_info = GetInput("Alpha"); - auto output_info = GetOutput("Out"); - - std::string slope_cast_name = slope_info[0].name; - if (slope_info[0].dtype == P2ODataType::FP64) { - slope_cast_name = helper_->AutoCast({slope_info[0].name}, P2ODataType::FP64, - P2ODataType::FP32); - } - - if (slope_info[0].Rank() != input_info[0].Rank()) { - Assert(slope_info[0].Rank() <= 1, - "Paddle2ONNX: Only support rank of alpha <= 1 while rank of alpha " - "is not equal with rank of input for operator prelu."); - Assert( - input_info[0].Rank() > 1, - "Paddle2ONNX: Rank of input should greater than 2 for operator prelu."); - std::vector shape_value(input_info[0].Rank() - 1, 1); - shape_value[0] = -1; - slope_cast_name = helper_->Reshape(slope_cast_name, shape_value); - } - - if (input_info[0].dtype == P2ODataType::FP64) { - std::string x_cast_name = helper_->AutoCast( - {input_info[0].name}, P2ODataType::FP64, P2ODataType::FP32); - auto node = helper_->MakeNode("PRelu", {x_cast_name, slope_cast_name}); - helper_->AutoCast(node->output(0), {output_info[0].name}, P2ODataType::FP32, - P2ODataType::FP64); - } else { - helper_->MakeNode("PRelu", {input_info[0].name, slope_cast_name}, - {output_info[0].name}); - } -} - -void SeluMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - auto node = - helper_->MakeNode("Selu", {input_info[0].name}, {output_info[0].name}); - AddAttribute(node, "alpha", alpha_); - AddAttribute(node, "gamma", scale_); -} - -void HardSigmoidMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - auto node = helper_->MakeNode("HardSigmoid", {input_info[0].name}, - {output_info[0].name}); - AddAttribute(node, "alpha", alpha_); - AddAttribute(node, "beta", beta_); -} - -void SwishMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - std::string beta_node = - helper_->Constant({}, GetOnnxDtype(input_info[0].dtype), beta_); - // TODO(jiangjiajun) eliminate multiply with a constant of value 1 - // TODO(jiangjiajun) eliminate add with a constant of value 0 - auto beta_x_node = helper_->MakeNode("Mul", {input_info[0].name, beta_node}); - auto sigmod_node = helper_->MakeNode("Sigmoid", {beta_x_node->output(0)}); - helper_->MakeNode("Mul", {input_info[0].name, sigmod_node->output(0)}, - {output_info[0].name}); -} - -void HardSwishMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - std::string scale_node = - helper_->Constant({}, GetOnnxDtype(input_info[0].dtype), scale_); - std::string offset_node = - helper_->Constant({}, GetOnnxDtype(input_info[0].dtype), offset_); - - auto add_node = helper_->MakeNode("Add", {input_info[0].name, offset_node}); - auto clip_node = - helper_->Clip(add_node->output(0), 0.0, threshold_, input_info[0].dtype); - - auto mul_node = helper_->MakeNode("Mul", {input_info[0].name, clip_node}); - helper_->MakeNode("Div", {mul_node->output(0), scale_node}, - {output_info[0].name}); -} - -void HardSwishMapper::Opset14() { - if (fabs(offset_ - 3.0) > 1e-05 || fabs(scale_ - 6.0) > 1e-05 || - fabs(threshold_ - 6.0) > 1e-05) { - return Opset7(); - } - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - helper_->MakeNode("HardSwish", {input_info[0].name}, {output_info[0].name}); -} - -void LeakyReluMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - auto node = helper_->MakeNode("LeakyRelu", {input_info[0].name}, - {output_info[0].name}); - AddAttribute(node, "alpha", alpha_); -} - -void GeluMapper::Opset9() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - auto input_onnx_dtype = GetOnnxDtype(input_info[0].dtype); - double sqrt_2_value = 1.4142135623730951; - double scale_value = 0.5; - double const_1_value = 1.0; - auto sqrt_2 = - helper_->Constant({}, ONNX_NAMESPACE::TensorProto::FLOAT, sqrt_2_value); - auto scale = - helper_->Constant({}, ONNX_NAMESPACE::TensorProto::FLOAT, scale_value); - auto const_1 = - helper_->Constant({}, ONNX_NAMESPACE::TensorProto::FLOAT, const_1_value); - - auto input_name = helper_->AutoCast(input_info[0].name, input_info[0].dtype, - P2ODataType::FP32); - - // the computation formula follows - // https://www.paddlepaddle.org.cn/documentation/docs/zh/api/paddle/nn/functional/gelu_cn.html#gelu - auto erf0 = helper_->MakeNode("Div", {input_name, sqrt_2}); - auto erf1 = helper_->MakeNode("Erf", {erf0->output(0)}); - auto gelu0 = helper_->MakeNode("Add", {erf1->output(0), const_1}); - auto gelu1 = helper_->MakeNode("Mul", {input_name, gelu0->output(0)}); - - if (input_info[0].dtype != P2ODataType::FP32) { - auto out = helper_->MakeNode("Mul", {gelu1->output(0), scale}); - auto cast_out = - helper_->MakeNode("Cast", {out->output(0)}, {output_info[0].name}); - AddAttribute(cast_out, "to", GetOnnxDtype(input_info[0].dtype)); - } else { - helper_->MakeNode("Mul", {gelu1->output(0), scale}, {output_info[0].name}); - } -} - -void SoftMaxMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - if (input_info[0].Rank() == 0) { - auto unsqueeze = helper_->Unsqueeze(input_info[0].name, {0}); - auto node = helper_->MakeNode("Softmax", {unsqueeze}); - AddAttribute(node, "axis", static_cast(0)); - helper_->Squeeze(node->output(0), output_info[0].name, {0}); - } else { - if (axis_ < 0) { - axis_ = axis_ + output_info[0].Rank(); - } - if (axis_ == output_info[0].Rank() - 1) { - auto node = helper_->MakeNode("Softmax", {input_info[0].name}, - {output_info[0].name}); - AddAttribute(node, "axis", axis_); - } else { - std::vector perm = Arange(0, output_info[0].Rank()); - perm[output_info[0].Rank() - 1] = axis_; - perm[axis_] = output_info[0].Rank() - 1; - auto transpose_node = - helper_->MakeNode("Transpose", {input_info[0].name}); - AddAttribute(transpose_node, "perm", perm); - auto softmax_node = - helper_->MakeNode("Softmax", {transpose_node->output(0)}); - int64_t axis_last = -1; - AddAttribute(softmax_node, "axis", axis_last); - auto transpose_node_last = helper_->MakeNode( - "Transpose", {softmax_node->output(0)}, {output_info[0].name}); - AddAttribute(transpose_node_last, "perm", perm); - } - } -} - -void SoftMaxMapper::Opset13() { - int64_t axis; - GetAttr("axis", &axis); - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - if (input_info[0].Rank() == 0) { - auto unsqueeze = helper_->Unsqueeze(input_info[0].name, {0}); - auto node = helper_->MakeNode("Softmax", {unsqueeze}); - AddAttribute(node, "axis", static_cast(0)); - helper_->Squeeze(node->output(0), output_info[0].name, {0}); - } else { - auto node = helper_->MakeNode("Softmax", {input_info[0].name}, - {output_info[0].name}); - AddAttribute(node, "axis", axis); - } -} - -void BReluMapper::Opset7() { - auto x_info = GetInput("X"); - helper_->Clip(x_info[0].name, GetOutput("Out")[0].name, t_min_, t_max_, - x_info[0].dtype); -} - -void EluMapper::Opset7() { - auto node = helper_->MakeNode("Elu", {GetInput("X")[0].name}, - {GetOutput("Out")[0].name}); - AddAttribute(node, "alpha", alpha_); -} - -void HardShrinkMapper::Opset9() { - auto node = helper_->MakeNode("Shrink", {GetInput("X")[0].name}, - {GetOutput("Out")[0].name}); - AddAttribute(node, "lambd", threshold_); - AddAttribute(node, "bias", float(0.0)); -} - -int32_t MishMapper::GetMinOpset(bool verbose) { - if (fabs(threshold_ - 20.0) > 1e-05) { - Error() << "Only support threshold = 20.0." << std::endl; - return -1; - } - return 7; -} - -void MishMapper::Opset7() { - auto input_info = GetInput("X"); - auto out_info = GetOutput("Out"); - auto input = helper_->AutoCast(input_info[0].name, input_info[0].dtype, - P2ODataType::FP32); - auto softplus = helper_->MakeNode("Softplus", {input})->output(0); - auto tanh = helper_->MakeNode("Tanh", {softplus})->output(0); - auto output = helper_->MakeNode("Mul", {input, tanh})->output(0); - helper_->AutoCast(output, out_info[0].name, P2ODataType::FP32, - out_info[0].dtype); -} - -void SquareMapper::Opset7() { - auto input_info = GetInput("X"); - helper_->MakeNode("Mul", {input_info[0].name, input_info[0].name}, - {GetOutput("Out")[0].name}); -} - -void SoftShrinkMapper::Opset9() { - auto node = helper_->MakeNode("Shrink", {GetInput("X")[0].name}, - {GetOutput("Out")[0].name}); - AddAttribute(node, "lambd", lambda_); - AddAttribute(node, "bias", lambda_); -} - -void SizeMapper::Opset7() { - auto out_info = GetOutput("Out"); - auto output = - helper_->MakeNode("Size", {GetInput("Input")[0].name})->output(0); - output = helper_->AutoCast(output, out_info[0].name, P2ODataType::INT64, - out_info[0].dtype); -} - -void RsqrtMapper::Opset7() { - auto output = helper_->MakeNode("Sqrt", {GetInput("X")[0].name})->output(0); - helper_->MakeNode("Reciprocal", {output}, {GetOutput("Out")[0].name}); -} - -void TanhShrinkMapper::Opset7() { - auto x_info = GetInput("X"); - auto tanh = helper_->MakeNode("Tanh", {x_info[0].name})->output(0); - helper_->MakeNode("Sub", {x_info[0].name, tanh}, {GetOutput("Out")[0].name}); -} - -void LogSigmoidMapper::Opset7() { - auto output = - helper_->MakeNode("Sigmoid", {GetInput("X")[0].name})->output(0); - helper_->MakeNode("Log", {output}, {GetOutput("Out")[0].name}); -} - -void LogSoftmaxMapper::Opset7() { - auto input_info = GetInput("X"); - auto axis = axis_; - if (input_info[0].Rank() == 0) { - auto unsqueeze = helper_->Unsqueeze(input_info[0].name, {0}); - auto node = helper_->MakeNode("LogSoftmax", {unsqueeze}); - AddAttribute(node, "axis", static_cast(0)); - helper_->Squeeze(node->output(0), GetOutput("Out")[0].name, {0}); - } else { - if (axis < 0) { - axis += input_info[0].Rank(); - } - if (axis == input_info[0].Rank() - 1) { - auto node = helper_->MakeNode("LogSoftmax", {input_info[0].name}, - {GetOutput("Out")[0].name}); - AddAttribute(node, "axis", axis); - } else { - auto perm = Arange(0, input_info[0].Rank()); - perm[input_info[0].Rank() - 1] = axis; - perm[axis] = input_info[0].Rank() - 1; - auto output = helper_->Transpose(input_info[0].name, perm); - auto node = helper_->MakeNode("LogSoftmax", {output}); - AddAttribute(node, "axis", int64_t(-1)); - helper_->Transpose(node->output(0), GetOutput("Out")[0].name, perm); - } - } -} - -void ThresholdedReluMapper::Opset10() { - auto x_info = GetInput("X"); - auto out_info = GetOutput("Out"); - auto input = x_info[0].name; - if (x_info[0].dtype != P2ODataType::FP32) { - input = helper_->AutoCast(input, x_info[0].dtype, P2ODataType::FP32); - auto node = helper_->MakeNode("ThresholdedRelu", {input}); - AddAttribute(node, "alpha", threshold_); - helper_->AutoCast(node->output(0), out_info[0].name, P2ODataType::FP32, - out_info[0].dtype); - } else { - auto node = - helper_->MakeNode("ThresholdedRelu", {input}, {out_info[0].name}); - AddAttribute(node, "alpha", threshold_); - } -} - -void Log1PMapper::Opset7() { - auto x_info = GetInput("X"); - auto out_info = GetOutput("Out"); - auto one = helper_->Constant({}, GetOnnxDtype(x_info[0].dtype), float(1.0)); - auto input = helper_->MakeNode("Add", {x_info[0].name, one})->output(0); - helper_->MakeNode("Log", {input}, {out_info[0].name}); -} - -void Log2Mapper::Opset7() { - auto x_info = GetInput("X"); - auto out_info = GetOutput("Out"); - double ln2 = 0.693147180559945309; - auto ln2_tensor = helper_->Constant({}, GetOnnxDtype(x_info[0].dtype), ln2); - auto output = helper_->MakeNode("Log", {x_info[0].name})->output(0); - helper_->MakeNode("Div", {output, ln2_tensor}, {out_info[0].name}); -} - -void Log10Mapper::Opset7() { - auto x_info = GetInput("X"); - auto out_info = GetOutput("Out"); - double ln10 = 2.30258509299404568401; - auto ln10_tensor = helper_->Constant({}, GetOnnxDtype(x_info[0].dtype), ln10); - auto output = helper_->MakeNode("Log", {x_info[0].name})->output(0); - helper_->MakeNode("Div", {output, ln10_tensor}, {out_info[0].name}); -} - -void SiluMapper::Opset7() { - auto x_info = GetInput("X"); - auto out_info = GetOutput("Out"); - auto out = helper_->MakeNode("Sigmoid", {x_info[0].name})->output(0); - helper_->MakeNode("Mul", {x_info[0].name, out}, {out_info[0].name}); -} -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/activation.h b/paddle2onnx/mapper/activation.h deleted file mode 100644 index d1143c5fbbd..00000000000 --- a/paddle2onnx/mapper/activation.h +++ /dev/null @@ -1,372 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. -#pragma once -#include -#include -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class ActivationMapper : public Mapper { - public: - ActivationMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - op_mapper_["relu"] = "Relu"; - op_mapper_["tanh"] = "Tanh"; - op_mapper_["log"] = "Log"; - op_mapper_["sigmoid"] = "Sigmoid"; - op_mapper_["sqrt"] = "Sqrt"; - op_mapper_["softplus"] = "Softplus"; - op_mapper_["exp"] = "Exp"; - op_mapper_["floor"] = "Floor"; - op_mapper_["cos"] = "Cos"; - op_mapper_["sin"] = "Sin"; - op_mapper_["round"] = "Round"; - op_mapper_["abs"] = "Abs"; - op_mapper_["acos"] = "Acos"; - op_mapper_["asin"] = "Asin"; - op_mapper_["atan"] = "Atan"; - op_mapper_["sinh"] = "Sinh"; - op_mapper_["tan"] = "Tan"; - op_mapper_["ceil"] = "Ceil"; - op_mapper_["cosh"] = "Cosh"; - op_mapper_["erf"] = "Erf"; - op_mapper_["sign"] = "Sign"; - op_mapper_["softsign"] = "Softsign"; - op_mapper_["reciprocal"] = "Reciprocal"; - } - - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - std::map op_mapper_; -}; - -class Relu6Mapper : public Mapper { - public: - Relu6Mapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("threshold", &threshold_); - } - - void Opset7(); - - private: - float threshold_; -}; - -class PReluMapper : public Mapper { - public: - PReluMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - - int32_t GetMinOpset(bool verbose = false); - void Opset7(); -}; - -class SeluMapper : public Mapper { - public: - SeluMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("alpha", &alpha_); - GetAttr("scale", &scale_); - } - - void Opset7(); - - private: - float alpha_; - float scale_; -}; - -class HardSigmoidMapper : public Mapper { - public: - HardSigmoidMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("slope", &alpha_); - GetAttr("offset", &beta_); - } - - void Opset7(); - - private: - float alpha_; - float beta_; -}; - -class SwishMapper : public Mapper { - public: - SwishMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("beta", &beta_); - } - - void Opset7(); - - private: - float beta_; -}; - -class HardSwishMapper : public Mapper { - public: - HardSwishMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("scale", &scale_); - GetAttr("offset", &offset_); - GetAttr("threshold", &threshold_); - } - - void Opset7(); - void Opset14(); - - private: - float scale_; - float offset_; - float threshold_; -}; - -class LeakyReluMapper : public Mapper { - public: - LeakyReluMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("alpha", &alpha_); - } - - void Opset7(); - - private: - float alpha_; -}; - -class GeluMapper : public Mapper { - public: - GeluMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - - int32_t GetMinOpset(bool verbose = false) { - Logger(verbose, 9) << RequireOpset(9) << std::endl; - return 9; - } - - void Opset9(); -}; - -class SoftMaxMapper : public Mapper { - public: - SoftMaxMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("axis", &axis_); - } - - void Opset7(); - void Opset13(); - - private: - int64_t axis_ = -1; -}; - -class BReluMapper : public Mapper { - public: - BReluMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("t_min", &t_min_); - GetAttr("t_max", &t_max_); - } - - void Opset7(); - - private: - float t_min_; - float t_max_; -}; - -class EluMapper : public Mapper { - public: - EluMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("alpha", &alpha_); - } - void Opset7(); - - private: - float alpha_; -}; - -class HardShrinkMapper : public Mapper { - public: - HardShrinkMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("threshold", &threshold_); - } - int32_t GetMinOpset(bool verbose = false) { - Logger(verbose, 9) << RequireOpset(9) << std::endl; - return 9; - } - void Opset9(); - - private: - float threshold_; -}; - -class MishMapper : public Mapper { - public: - MishMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("threshold", &threshold_); - } - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - float threshold_; -}; - -class SquareMapper : public Mapper { - public: - SquareMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); -}; - -class SizeMapper : public Mapper { - public: - SizeMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); -}; - -class LogSigmoidMapper : public Mapper { - public: - LogSigmoidMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); -}; - -class RsqrtMapper : public Mapper { - public: - RsqrtMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); -}; - -class LogSoftmaxMapper : public Mapper { - public: - LogSoftmaxMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("axis", &axis_); - } - void Opset7(); - - private: - int64_t axis_; -}; - -class SoftShrinkMapper : public Mapper { - public: - SoftShrinkMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("lambda", &lambda_); - } - int32_t GetMinOpset(bool verbose = false) { - Logger(verbose, 9) << RequireOpset(9) << std::endl; - return 9; - } - void Opset9(); - - private: - float lambda_; -}; - -class ThresholdedReluMapper : public Mapper { - public: - ThresholdedReluMapper(const PaddleParser& p, OnnxHelper* helper, - int64_t block_id, int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("threshold", &threshold_); - } - int32_t GetMinOpset(bool verbose = false) { - Logger(verbose, 10) << RequireOpset(10) << std::endl; - return 10; - } - void Opset10(); - - private: - float threshold_; -}; - -class TanhShrinkMapper : public Mapper { - public: - TanhShrinkMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); -}; - -class Log1PMapper : public Mapper { - public: - Log1PMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); -}; - -class Log2Mapper : public Mapper { - public: - Log2Mapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); -}; - -class Log10Mapper : public Mapper { - public: - Log10Mapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); -}; - -class SiluMapper : public Mapper { - public: - SiluMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/data_helper.h b/paddle2onnx/mapper/data_helper.h deleted file mode 100644 index df488194b7e..00000000000 --- a/paddle2onnx/mapper/data_helper.h +++ /dev/null @@ -1,30 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include "paddle2onnx/utils/utils.h" - -namespace paddle2onnx { - -inline std::vector Arange(int64_t start, int64_t end) { - Assert(end > start, "In arrange(), end must be greater than start."); - std::vector res; - res.resize(end - start); - for (auto i = start; i < end; ++i) { - res[i - start] = i; - } - return res; -} -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/detection/multiclass_nms.cc b/paddle2onnx/mapper/detection/multiclass_nms.cc deleted file mode 100644 index 3c262b6b88a..00000000000 --- a/paddle2onnx/mapper/detection/multiclass_nms.cc +++ /dev/null @@ -1,360 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/detection/multiclass_nms.h" - -namespace paddle2onnx { - -REGISTER_MAPPER(multiclass_nms3, NMSMapper); - -int32_t NMSMapper::GetMinOpset(bool verbose) { - auto boxes_info = GetInput("BBoxes"); - auto score_info = GetInput("Scores"); - if (score_info[0].Rank() != 3) { - Error() << "Lod Tensor input is not supported, which means the shape of " - "input(scores) is [M, C] now, but Paddle2ONNX only support [N, " - "C, M]." - << std::endl; - return -1; - } - if (boxes_info[0].Rank() != 3) { - Error() << "Only support input boxes as 3-D Tensor, but now it's rank is " - << boxes_info[0].Rank() << "." << std::endl; - return -1; - } - if (score_info[0].shape[1] <= 0) { - Error() << "The 2nd-dimension of score should be fixed(means the number of " - "classes), but now it's " - << score_info[0].shape[1] << "." << std::endl; - return -1; - } - - if (export_as_custom_op || this->deploy_backend == "tensorrt") { - return 7; - } - - Logger(verbose, 10) << RequireOpset(10) << std::endl; - return 10; -} - -void NMSMapper::KeepTopK(const std::string& selected_indices) { - auto boxes_info = GetInput("BBoxes"); - auto score_info = GetInput("Scores"); - auto out_info = GetOutput("Out"); - auto index_info = GetOutput("Index"); - auto num_rois_info = GetOutput("NmsRoisNum"); - auto value_0 = - helper_->Constant({1}, ONNX_NAMESPACE::TensorProto::INT64, int64_t(0)); - auto value_1 = - helper_->Constant({1}, ONNX_NAMESPACE::TensorProto::INT64, int64_t(1)); - auto value_2 = - helper_->Constant({1}, ONNX_NAMESPACE::TensorProto::INT64, int64_t(2)); - auto value_neg_1 = - helper_->Constant({1}, ONNX_NAMESPACE::TensorProto::INT64, int64_t(-1)); - - auto class_id = helper_->MakeNode("Gather", {selected_indices, value_1}); - AddAttribute(class_id, "axis", int64_t(1)); - - auto box_id = helper_->MakeNode("Gather", {selected_indices, value_2}); - AddAttribute(box_id, "axis", int64_t(1)); - - auto filtered_class_id = class_id->output(0); - auto filtered_box_id = box_id->output(0); - if (background_label_ >= 0) { - auto filter_indices = MapperHelper::Get()->GenName("nms.filter_background"); - auto squeezed_class_id = - helper_->Squeeze(class_id->output(0), std::vector(1, 1)); - if (background_label_ > 0) { - auto background = helper_->Constant( - {1}, ONNX_NAMESPACE::TensorProto::INT64, background_label_); - auto diff = helper_->MakeNode("Sub", {squeezed_class_id, background}); - helper_->MakeNode("NonZero", {diff->output(0)}, {filter_indices}); - } else if (background_label_ == 0) { - helper_->MakeNode("NonZero", {squeezed_class_id}, {filter_indices}); - } - auto new_class_id = - helper_->MakeNode("Gather", {filtered_class_id, filter_indices}); - AddAttribute(new_class_id, "axis", int64_t(0)); - auto new_box_id = - helper_->MakeNode("Gather", {box_id->output(0), filter_indices}); - AddAttribute(new_box_id, "axis", int64_t(0)); - filtered_class_id = new_class_id->output(0); - filtered_box_id = new_box_id->output(0); - } - - // Here is a little complicated - // Since we need to gather all the scores for the final boxes to filter the - // top-k boxes Now we have the follow inputs - // - scores: [N, C, M] N means batch size(but now it will be regarded as - // 1); C means number of classes; M means number of boxes for each classes - // - selected_indices: [num_selected_indices, 3], and 3 means [batch, - // class_id, box_id]. We will use this inputs to gather score - // So now we will first flatten `scores` to shape of [1 * C * M], then we - // gather scores by each elements in `selected_indices` The index need be - // calculated as - // `gather_index = class_id * M + box_id` - auto flatten_score = helper_->Flatten(score_info[0].name); - auto num_boxes_each_class = helper_->Constant( - {1}, ONNX_NAMESPACE::TensorProto::INT64, score_info[0].shape[2]); - auto gather_indices_0 = - helper_->MakeNode("Mul", {filtered_class_id, num_boxes_each_class}); - auto gather_indices_1 = - helper_->MakeNode("Add", {gather_indices_0->output(0), filtered_box_id}); - auto gather_indices = helper_->Flatten(gather_indices_1->output(0)); - auto gathered_scores = - helper_->MakeNode("Gather", {flatten_score, gather_indices}); - AddAttribute(gathered_scores, "axis", int64_t(0)); - - // Now we will perform keep_top_k process - // First we need to check if the number of remaining boxes is greater than - // keep_top_k Otherwise, we will downgrade the keep_top_k to number of - // remaining boxes - auto final_classes = filtered_class_id; - auto final_boxes_id = filtered_box_id; - auto final_scores = gathered_scores->output(0); - if (keep_top_k_ > 0) { - // get proper topk - auto shape_of_scores = helper_->MakeNode("Shape", {final_scores}); - auto num_of_boxes = - helper_->Slice(shape_of_scores->output(0), std::vector(1, 0), - std::vector(1, 0), std::vector(1, 1)); - auto top_k = - helper_->Constant({1}, ONNX_NAMESPACE::TensorProto::INT64, keep_top_k_); - auto ensemble_value = helper_->MakeNode("Concat", {num_of_boxes, top_k}); - AddAttribute(ensemble_value, "axis", int64_t(0)); - auto new_top_k = - helper_->MakeNode("ReduceMin", {ensemble_value->output(0)}); - AddAttribute(new_top_k, "axes", std::vector(1, 0)); - AddAttribute(new_top_k, "keepdims", int64_t(1)); - - // the output is topk_scores, topk_score_indices - auto topk_node = - helper_->MakeNode("TopK", {final_scores, new_top_k->output(0)}, 2); - auto topk_scores = - helper_->MakeNode("Gather", {final_scores, topk_node->output(1)}); - AddAttribute(topk_scores, "axis", int64_t(0)); - filtered_class_id = - helper_->MakeNode("Squeeze", {filtered_class_id})->output(0); - auto topk_classes = - helper_->MakeNode("Gather", {filtered_class_id, topk_node->output(1)}); - AddAttribute(topk_classes, "axis", int64_t(0)); - filtered_box_id = - helper_->MakeNode("Squeeze", {filtered_box_id})->output(0); - auto topk_boxes_id = - helper_->MakeNode("Gather", {filtered_box_id, topk_node->output(1)}); - AddAttribute(topk_boxes_id, "axis", int64_t(0)); - - final_boxes_id = topk_boxes_id->output(0); - final_scores = topk_scores->output(0); - final_classes = topk_classes->output(0); - } - - auto flatten_boxes_id = helper_->Flatten({final_boxes_id}); - auto gathered_selected_boxes = - helper_->MakeNode("Gather", {boxes_info[0].name, flatten_boxes_id}); - AddAttribute(gathered_selected_boxes, "axis", int64_t(1)); - - auto float_classes = helper_->MakeNode("Cast", {final_classes}); - AddAttribute(float_classes, "to", ONNX_NAMESPACE::TensorProto::FLOAT); - - std::vector shape{1, -1, 1}; - auto unsqueezed_scores = helper_->Reshape({final_scores}, shape); - - auto unsqueezed_class = helper_->Reshape({float_classes->output(0)}, shape); - - auto box_result = - helper_->MakeNode("Concat", {unsqueezed_class, unsqueezed_scores, - gathered_selected_boxes->output(0)}); - AddAttribute(box_result, "axis", int64_t(2)); - helper_->Squeeze({box_result->output(0)}, {out_info[0].name}, - std::vector(1, 0)); - - // other outputs, we don't use sometimes - // there's lots of Cast in exporting - // TODO(jiangjiajun) A pass to eleminate all the useless Cast is needed - auto reshaped_index_result = - helper_->Reshape({flatten_boxes_id}, {int64_t(-1), int64_t(1)}); - auto index_result = - helper_->MakeNode("Cast", {reshaped_index_result}, {index_info[0].name}); - AddAttribute(index_result, "to", GetOnnxDtype(index_info[0].dtype)); - - auto out_box_shape = helper_->MakeNode("Shape", {out_info[0].name}); - auto num_rois_result = - helper_->Slice({out_box_shape->output(0)}, std::vector(1, 0), - std::vector(1, 0), std::vector(1, 1)); - auto int32_num_rois_result = - helper_->AutoCast(num_rois_result, num_rois_info[0].name, - P2ODataType::INT64, num_rois_info[0].dtype); -} - -void NMSMapper::Opset10() { - if (this->deploy_backend == "tensorrt") { - return ExportForTensorRT(); - } - auto boxes_info = GetInput("BBoxes"); - auto score_info = GetInput("Scores"); - if (boxes_info[0].shape[0] != 1) { - Warn() - << "[WARNING] Due to the operator multiclass_nms3, the exported ONNX " - "model will only supports inference with input batch_size == 1." - << std::endl; - } - int64_t num_classes = score_info[0].shape[1]; - auto score_threshold = helper_->Constant( - {1}, ONNX_NAMESPACE::TensorProto::FLOAT, score_threshold_); - auto nms_threshold = helper_->Constant( - {1}, ONNX_NAMESPACE::TensorProto::FLOAT, nms_threshold_); - auto nms_top_k = - helper_->Constant({1}, ONNX_NAMESPACE::TensorProto::INT64, nms_top_k_); - - auto selected_box_index = MapperHelper::Get()->GenName("nms.selected_index"); - if (normalized_) { - helper_->MakeNode("NonMaxSuppression", - {boxes_info[0].name, score_info[0].name, nms_top_k, - nms_threshold, score_threshold}, - {selected_box_index}); - } else { - auto value_1 = - helper_->Constant({1}, GetOnnxDtype(boxes_info[0].dtype), float(1.0)); - auto split_boxes = helper_->Split(boxes_info[0].name, - std::vector(4, 1), int64_t(2)); - auto xmax = helper_->MakeNode("Add", {split_boxes[2], value_1}); - auto ymax = helper_->MakeNode("Add", {split_boxes[3], value_1}); - auto new_boxes = helper_->MakeNode( - "Concat", - {split_boxes[0], split_boxes[1], xmax->output(0), ymax->output(0)}); - AddAttribute(new_boxes, "axis", int64_t(2)); - helper_->MakeNode("NonMaxSuppression", - {new_boxes->output(0), score_info[0].name, nms_top_k, - nms_threshold, score_threshold}, - {selected_box_index}); - } - KeepTopK(selected_box_index); -} - -void NMSMapper::ExportAsCustomOp() { - auto boxes_info = GetInput("BBoxes"); - auto score_info = GetInput("Scores"); - auto out_info = GetOutput("Out"); - auto index_info = GetOutput("Index"); - auto num_rois_info = GetOutput("NmsRoisNum"); - auto node = helper_->MakeNode( - custom_op_name, {boxes_info[0].name, score_info[0].name}, - {out_info[0].name, index_info[0].name, num_rois_info[0].name}); - node->set_domain("Paddle"); - int64_t normalized = normalized_ ? 1 : 0; - AddAttribute(node, "normalized", normalized); - AddAttribute(node, "nms_threshold", nms_threshold_); - AddAttribute(node, "score_threshold", score_threshold_); - AddAttribute(node, "nms_eta", nms_eta_); - AddAttribute(node, "nms_top_k", nms_top_k_); - AddAttribute(node, "background_label", background_label_); - AddAttribute(node, "keep_top_k", keep_top_k_); - helper_->MakeValueInfo(boxes_info[0].name, boxes_info[0].dtype, - boxes_info[0].shape); - helper_->MakeValueInfo(score_info[0].name, score_info[0].dtype, - score_info[0].shape); - helper_->MakeValueInfo(out_info[0].name, out_info[0].dtype, - out_info[0].shape); - helper_->MakeValueInfo(index_info[0].name, index_info[0].dtype, - index_info[0].shape); - helper_->MakeValueInfo(num_rois_info[0].name, num_rois_info[0].dtype, - num_rois_info[0].shape); -} - -void NMSMapper::ExportForTensorRT() { - auto boxes_info = GetInput("BBoxes"); - auto score_info = GetInput("Scores"); - auto out_info = GetOutput("Out"); - auto index_info = GetOutput("Index"); - auto num_rois_info = GetOutput("NmsRoisNum"); - - auto scores = helper_->Transpose(score_info[0].name, {0, 2, 1}); - auto boxes = helper_->Unsqueeze(boxes_info[0].name, {2}); - int64_t num_classes = score_info[0].shape[1]; - auto repeats = - helper_->Constant(GetOnnxDtype(P2ODataType::INT64), - std::vector({1, 1, num_classes, 1})); - boxes = helper_->MakeNode("Tile", {boxes, repeats})->output(0); - - auto nms_node = - helper_->MakeNode("BatchedNMSDynamic_TRT", {boxes, scores}, 4); - AddAttribute(nms_node, "shareLocation", int64_t(0)); - AddAttribute(nms_node, "backgroundLabelId", background_label_); - AddAttribute(nms_node, "numClasses", num_classes); - int64_t nms_top_k = nms_top_k_; - int64_t keep_top_k = keep_top_k_; - if (nms_top_k > 4096) { - Warn() - << "Paramter nms_top_k:" << nms_top_k - << " is exceed limit in TensorRT BatchedNMS plugin, will force to 4096." - << std::endl; - nms_top_k = 4096; - } - if (keep_top_k > 4096) { - Warn() - << "Parameter keep_top_k:" << keep_top_k - << " is exceed limit in TensorRT BatchedNMS plugin, will force to 4096." - << std::endl; - keep_top_k = 4096; - } - AddAttribute(nms_node, "topK", nms_top_k); - AddAttribute(nms_node, "keepTopK", keep_top_k); - AddAttribute(nms_node, "scoreThreshold", score_threshold_); - AddAttribute(nms_node, "iouThreshold", nms_threshold_); - if (normalized_) { - AddAttribute(nms_node, "isNormalized", int64_t(1)); - } else { - AddAttribute(nms_node, "isNormalized", int64_t(0)); - } - AddAttribute(nms_node, "clipBoxes", int64_t(0)); - nms_node->set_domain("Paddle"); - - auto num_rois = helper_->Reshape(nms_node->output(0), {-1}); - helper_->AutoCast(num_rois, num_rois_info[0].name, P2ODataType::INT32, - num_rois_info[0].dtype); - - auto out_classes = helper_->Reshape(nms_node->output(3), {-1, 1}); - auto out_scores = helper_->Reshape(nms_node->output(2), {-1, 1}); - auto out_boxes = helper_->Reshape(nms_node->output(1), {-1, 4}); - out_classes = - helper_->AutoCast(out_classes, P2ODataType::INT32, P2ODataType::FP32); - helper_->Concat({out_classes, out_scores, out_boxes}, {out_info[0].name}, 1); - - // EfficientNMS_TRT cannot get the same result, so disable now - // auto nms_node = helper_->MakeNode("EfficientNMS_TRT", {boxes_info[0].name, - // score}, 4); - // AddAttribute(nms_node, "plugin_version", "1"); - // AddAttribute(nms_node, "background_class", background_label_); - // AddAttribute(nms_node, "max_output_boxes", nms_top_k_); - // AddAttribute(nms_node, "score_threshold", score_threshold_); - // AddAttribute(nms_node, "iou_threshold", nms_threshold_); - // AddAttribute(nms_node, "score_activation", int64_t(0)); - // AddAttribute(nms_node, "box_coding", int64_t(0)); - // nms_node->set_domain("Paddle"); - // - // auto num_rois = helper_->Reshape(nms_node->output(0), {-1}); - // helper_->AutoCast(num_rois, num_rois_info[0].name, P2ODataType::INT32, - // num_rois_info[0].dtype); - // - // auto out_classes = helper_->Reshape(nms_node->output(3), {-1, 1}); - // auto out_scores = helper_->Reshape(nms_node->output(2), {-1, 1}); - // auto out_boxes = helper_->Reshape(nms_node->output(1), {-1, 4}); - // out_classes = helper_->AutoCast(out_classes, P2ODataType::INT32, - // P2ODataType::FP32); - // helper_->Concat({out_classes, out_scores, out_boxes}, {out_info[0].name}, - // 1); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/detection/multiclass_nms.h b/paddle2onnx/mapper/detection/multiclass_nms.h deleted file mode 100644 index 84711a5ad22..00000000000 --- a/paddle2onnx/mapper/detection/multiclass_nms.h +++ /dev/null @@ -1,62 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class NMSMapper : public Mapper { - public: - NMSMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - // NMS is a post process operators for object detection - // We have found there're difference between `multi_class_nms3` in - // PaddlePaddle and `NonMaxSuppresion` in ONNX - MarkAsExperimentalOp(); - GetAttr("normalized", &normalized_); - GetAttr("nms_threshold", &nms_threshold_); - GetAttr("score_threshold", &score_threshold_); - GetAttr("nms_eta", &nms_eta_); - // The `nms_top_k` in Paddle and `max_output_boxes_per_class` in ONNX share - // the same meaning But the filter process may not be same Since NMS is just - // a post process for Detection, we are not going to export it with exactly - // same result. We will make a precision performance in COCO or Pascal VOC - // data later. - GetAttr("nms_top_k", &nms_top_k_); - GetAttr("background_label", &background_label_); - GetAttr("keep_top_k", &keep_top_k_); - } - - int32_t GetMinOpset(bool verbose = false); - void KeepTopK(const std::string& selected_indices); - void Opset10(); - void ExportForTensorRT(); - void ExportAsCustomOp(); - - private: - bool normalized_; - float nms_threshold_; - float score_threshold_; - float nms_eta_; - int64_t nms_top_k_; - int64_t background_label_; - int64_t keep_top_k_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/detection/roi_align.cc b/paddle2onnx/mapper/detection/roi_align.cc deleted file mode 100755 index 88629a28190..00000000000 --- a/paddle2onnx/mapper/detection/roi_align.cc +++ /dev/null @@ -1,43 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/detection/roi_align.h" - -namespace paddle2onnx { -REGISTER_MAPPER(roi_align, RoiAlignMapper) - -void RoiAlignMapper::Opset10() { - auto x_info = GetInput("X"); - auto rois_info = GetInput("ROIs"); - auto out_info = GetOutput("Out"); - - auto roi_shape = helper_->MakeNode("Shape", {rois_info[0].name})->output(0); - auto num_rois = - helper_->Slice(roi_shape, std::vector(1, 0), - std::vector(1, 0), std::vector(1, 1)); - auto value_zero = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, - std::vector(1, 0)); - auto batch_indices = - helper_->MakeNode("Expand", {value_zero, num_rois})->output(0); - auto roi_align_node = helper_->MakeNode( - "RoiAlign", {x_info[0].name, rois_info[0].name, batch_indices}, - {out_info[0].name}); - AddAttribute(roi_align_node, "output_height", pooled_height_); - AddAttribute(roi_align_node, "output_width", pooled_width_); - AddAttribute(roi_align_node, "sampling_ratio", sampling_ratio_); - AddAttribute(roi_align_node, "spatial_scale", spatial_scale_); - AddAttribute(roi_align_node, "mode", "avg"); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/detection/roi_align.h b/paddle2onnx/mapper/detection/roi_align.h deleted file mode 100755 index 68e2c27f92c..00000000000 --- a/paddle2onnx/mapper/detection/roi_align.h +++ /dev/null @@ -1,47 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class RoiAlignMapper : public Mapper { - public: - RoiAlignMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - MarkAsExperimentalOp(); - GetAttr("pooled_height", &pooled_height_); - GetAttr("pooled_width", &pooled_width_); - GetAttr("spatial_scale", &spatial_scale_); - GetAttr("sampling_ratio", &sampling_ratio_); - GetAttr("aligned", &aligned_); - } - - int32_t GetMinOpset(bool verbose = false) { - Logger(verbose, 10) << RequireOpset(10) << std::endl; - return 10; - } - void Opset10(); - - private: - int64_t pooled_height_; - int64_t pooled_width_; - float spatial_scale_; - int64_t sampling_ratio_; - bool aligned_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/detection/yolo_box.cc b/paddle2onnx/mapper/detection/yolo_box.cc deleted file mode 100644 index e9f92898901..00000000000 --- a/paddle2onnx/mapper/detection/yolo_box.cc +++ /dev/null @@ -1,254 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/detection/yolo_box.h" - -namespace paddle2onnx { - -REGISTER_MAPPER(yolo_box, YoloBoxMapper) - -int32_t YoloBoxMapper::GetMinOpset(bool verbose) { - Logger(verbose, 11) << RequireOpset(11) << std::endl; - return 11; -} - -void YoloBoxMapper::Opset11() { - auto x_info_ori = GetInput("X"); - - // handle the float64 input - auto x_info = x_info_ori; - if (x_info_ori[0].dtype != P2ODataType::FP32) { - x_info[0].name = helper_->AutoCast(x_info_ori[0].name, x_info_ori[0].dtype, - P2ODataType::FP32); - x_info[0].dtype = P2ODataType::FP32; - } - - auto im_size_info = GetInput("ImgSize"); - auto boxes_info = GetOutput("Boxes"); - auto scores_info = GetOutput("Scores"); - int64_t max_int = 999999; - - int64_t anchor_num = anchors_.size() / 2; - - auto x_shape = helper_->MakeNode("Shape", {x_info[0].name}); - std::vector nchw = helper_->Split( - x_shape->output(0), std::vector(4, 1), int64_t(0)); - std::string float_h = - helper_->AutoCast(nchw[2], P2ODataType::INT64, x_info[0].dtype); - std::string float_w = - helper_->AutoCast(nchw[3], P2ODataType::INT64, x_info[0].dtype); - - auto anchor_num_tensor = - helper_->Constant({1}, ONNX_NAMESPACE::TensorProto::INT64, anchor_num); - - auto x_name = x_info[0].name; - if (iou_aware_) { - // Here we use the feature that while value is very large, it equals to the - // ends This is a standared definition in ONNX However not sure all the - // inference engines implements `Slice` this way Let's handle this issue - // later - x_name = helper_->Slice(x_name, {0, 1, 2, 3}, {0, 0, 0, 0}, - {max_int, anchor_num, max_int, max_int}); - } - - auto unknown_dim = - helper_->Constant({1}, ONNX_NAMESPACE::TensorProto::INT64, int64_t(-1)); - auto shape_0 = helper_->MakeNode( - "Concat", {nchw[0], anchor_num_tensor, unknown_dim, nchw[2], nchw[3]}); - AddAttribute(shape_0, "axis", int64_t(0)); - auto reshaped_x = helper_->MakeNode("Reshape", {x_name, shape_0->output(0)}); - auto transposed_x = helper_->MakeNode("Transpose", {reshaped_x->output(0)}); - { - std::vector perm({0, 1, 3, 4, 2}); - AddAttribute(transposed_x, "perm", perm); - } - - // grid_x = np.tile(np.arange(w).reshape((1, w)), (h, 1)) - // grid_y = np.tile(np.arange(h).reshape((h, 1)), (1, w)) - auto float_value_0 = - helper_->Constant({}, GetOnnxDtype(x_info[0].dtype), float(0.0)); - auto float_value_1 = - helper_->Constant({}, GetOnnxDtype(x_info[0].dtype), float(1.0)); - auto scalar_float_w = helper_->Squeeze(float_w, {}); - auto scalar_float_h = helper_->Squeeze(float_h, {}); - auto grid_x_0 = helper_->MakeNode( - "Range", {float_value_0, scalar_float_w, float_value_1}); // shape is [w] - auto grid_y_0 = helper_->MakeNode( - "Range", {float_value_0, scalar_float_h, float_value_1}); // shape is [h] - auto grid_x_1 = helper_->MakeNode( - "Tile", {grid_x_0->output(0), nchw[2]}); // shape is [w*h] - auto grid_y_1 = helper_->MakeNode( - "Tile", {grid_y_0->output(0), nchw[3]}); // shape is [h*w] - auto int_value_1 = - helper_->Constant({1}, ONNX_NAMESPACE::TensorProto::INT64, float(1.0)); - auto grid_shape_x = - helper_->MakeNode("Concat", {nchw[2], nchw[3], int_value_1}); - auto grid_shape_y = - helper_->MakeNode("Concat", {nchw[3], nchw[2], int_value_1}); - AddAttribute(grid_shape_x, "axis", int64_t(0)); - AddAttribute(grid_shape_y, "axis", int64_t(0)); - auto grid_x = helper_->MakeNode( - "Reshape", {grid_x_1->output(0), grid_shape_x->output(0)}); - auto grid_y_2 = helper_->MakeNode( - "Reshape", {grid_y_1->output(0), grid_shape_y->output(0)}); - auto grid_y = helper_->MakeNode("Transpose", {grid_y_2->output(0)}); - { - std::vector perm({1, 0, 2}); - AddAttribute(grid_y, "perm", perm); - } - - auto grid = - helper_->MakeNode("Concat", {grid_x->output(0), grid_y->output(0)}); - AddAttribute(grid, "axis", int64_t(2)); - - // pred_box[:, :, :, :, 0] = (grid_x + sigmoid(pred_box[:, :, :, :, 0]) * - // scale_x_y + bias_x_y) / w pred_box[:, :, :, :, 1] = (grid_y + - // sigmoid(pred_box[:, :, :, :, 1]) * scale_x_y + bias_x_y) / h - auto pred_box_xy = - helper_->Slice(transposed_x->output(0), {0, 1, 2, 3, 4}, {0, 0, 0, 0, 0}, - {max_int, max_int, max_int, max_int, 2}); - auto scale_x_y = - helper_->Constant({1}, GetOnnxDtype(x_info[0].dtype), scale_x_y_); - float bias_x_y_value = (1.0 - scale_x_y_) / 2.0; - auto bias_x_y = - helper_->Constant({1}, GetOnnxDtype(x_info[0].dtype), bias_x_y_value); - auto wh = helper_->MakeNode("Concat", {float_w, float_h}); - AddAttribute(wh, "axis", int64_t(0)); - pred_box_xy = helper_->MakeNode("Sigmoid", {pred_box_xy})->output(0); - pred_box_xy = helper_->MakeNode("Mul", {pred_box_xy, scale_x_y})->output(0); - pred_box_xy = helper_->MakeNode("Add", {pred_box_xy, bias_x_y})->output(0); - pred_box_xy = - helper_->MakeNode("Add", {pred_box_xy, grid->output(0)})->output(0); - pred_box_xy = - helper_->MakeNode("Div", {pred_box_xy, wh->output(0)})->output(0); - - // anchors = [(anchors[i], anchors[i + 1]) for i in range(0, len(anchors), 2)] - // anchors_s = np.array( - // [(an_w / input_w, an_h / input_h) for an_w, an_h in anchors]) - // anchor_w = anchors_s[:, 0:1].reshape((1, an_num, 1, 1)) - // anchor_h = anchors_s[:, 1:2].reshape((1, an_num, 1, 1)) - std::vector valid_anchors(anchor_num); - valid_anchors.assign(anchors_.begin(), anchors_.begin() + anchor_num * 2); - auto anchors = - helper_->Constant(GetOnnxDtype(x_info[0].dtype), valid_anchors); - anchors = helper_->Reshape(anchors, {anchor_num, 2}); - - auto downsample = - helper_->Constant({1}, GetOnnxDtype(x_info[0].dtype), downsample_ratio_); - auto ori_wh = - helper_->MakeNode("Mul", {wh->output(0), downsample})->output(0); - anchors = helper_->MakeNode("Div", {anchors, ori_wh})->output(0); - // Following divide operation requires undirectional broadcast - // It satisfies the definition of ONNX, but now sure all the inference engines - // support this rule e.g TensorRT、OpenVINO anchor_w = anchors_s[:, - // 0:1].reshape((1, an_num, 1, 1)) anchor_h = anchors_s[:, 1:2].reshape((1, - // an_num, 1, 1)) pred_box[:, :, :, :, 2] = np.exp(pred_box[:, :, :, :, 2]) * - // anchor_w pred_box[:, :, :, :, 3] = np.exp(pred_box[:, :, :, :, 3]) * - // anchor_h - anchors = helper_->Reshape(anchors, {1, anchor_num, 1, 1, 2}); - auto pred_box_wh = - helper_->Slice(transposed_x->output(0), {0, 1, 2, 3, 4}, {0, 0, 0, 0, 2}, - {max_int, max_int, max_int, max_int, 4}); - pred_box_wh = helper_->MakeNode("Exp", {pred_box_wh})->output(0); - pred_box_wh = helper_->MakeNode("Mul", {pred_box_wh, anchors})->output(0); - - // if iou_aware: - // pred_conf = sigmoid(x[:, :, :, :, 4:5])**( - // 1 - iou_aware_factor) * sigmoid(ioup)**iou_aware_factor - // else: - // pred_conf = sigmoid(x[:, :, :, :, 4:5]) - auto confidence = - helper_->Slice(transposed_x->output(0), {0, 1, 2, 3, 4}, {0, 0, 0, 0, 4}, - {max_int, max_int, max_int, max_int, 5}); - std::string pred_conf = helper_->MakeNode("Sigmoid", {confidence})->output(0); - if (iou_aware_) { - auto ioup = helper_->Slice(x_info[0].name, {0, 1, 2, 3}, {0, 0, 0, 0}, - {max_int, anchor_num, max_int, max_int}); - ioup = helper_->Unsqueeze(ioup, {4}); - ioup = helper_->MakeNode("Sigmoid", {ioup})->output(0); - float power_value_0 = 1 - iou_aware_factor_; - auto power_0 = - helper_->Constant({1}, GetOnnxDtype(x_info[0].dtype), power_value_0); - auto power_1 = helper_->Constant({1}, GetOnnxDtype(x_info[0].dtype), - iou_aware_factor_); - ioup = helper_->MakeNode("Pow", {ioup, power_1})->output(0); - pred_conf = helper_->MakeNode("Pow", {pred_conf, power_0})->output(0); - pred_conf = helper_->MakeNode("Mul", {pred_conf, ioup})->output(0); - } - - // pred_conf[pred_conf < conf_thresh] = 0. - // pred_score = sigmoid(x[:, :, :, :, 5:]) * pred_conf - // pred_box = pred_box * (pred_conf > 0.).astype('float32') - auto value_2 = - helper_->Constant({1}, GetOnnxDtype(x_info[0].dtype), float(2.0)); - auto center = helper_->MakeNode("Div", {pred_box_wh, value_2})->output(0); - auto min_xy = helper_->MakeNode("Sub", {pred_box_xy, center})->output(0); - auto max_xy = helper_->MakeNode("Add", {pred_box_xy, center})->output(0); - - auto conf_thresh = - helper_->Constant({1}, GetOnnxDtype(x_info[0].dtype), conf_thresh_); - auto filter = - helper_->MakeNode("Greater", {pred_conf, conf_thresh})->output(0); - filter = helper_->AutoCast(filter, P2ODataType::BOOL, x_info[0].dtype); - pred_conf = helper_->MakeNode("Mul", {pred_conf, filter})->output(0); - auto pred_score = - helper_->Slice(transposed_x->output(0), {0, 1, 2, 3, 4}, {0, 0, 0, 0, 5}, - {max_int, max_int, max_int, max_int, max_int}); - pred_score = helper_->MakeNode("Sigmoid", {pred_score})->output(0); - pred_score = helper_->MakeNode("Mul", {pred_score, pred_conf})->output(0); - auto pred_box = helper_->Concat({min_xy, max_xy}, 4); - pred_box = helper_->MakeNode("Mul", {pred_box, filter})->output(0); - - auto value_neg_1 = - helper_->Constant({1}, ONNX_NAMESPACE::TensorProto::INT64, int64_t(-1)); - auto value_4 = - helper_->Constant({1}, ONNX_NAMESPACE::TensorProto::INT64, int64_t(4)); - auto new_shape = helper_->Concat({nchw[0], value_neg_1, value_4}, 0); - pred_box = helper_->MakeNode("Reshape", {pred_box, new_shape})->output(0); - - auto float_img_size = helper_->AutoCast( - im_size_info[0].name, im_size_info[0].dtype, x_info[0].dtype); - float_img_size = helper_->Unsqueeze(float_img_size, {1}); - auto split_im_hw = helper_->Split(float_img_size, {1, 1}, 2); - auto im_whwh = helper_->Concat( - {split_im_hw[1], split_im_hw[0], split_im_hw[1], split_im_hw[0]}, 2); - - if (!clip_bbox_) { - auto out = helper_->MakeNode("Mul", {pred_box, im_whwh})->output(0); - helper_->AutoCast(out, boxes_info[0].name, x_info[0].dtype, - boxes_info[0].dtype); - } else { - pred_box = helper_->MakeNode("Mul", {pred_box, im_whwh})->output(0); - auto im_wh = helper_->Concat({split_im_hw[1], split_im_hw[0]}, 2); - im_wh = helper_->MakeNode("Sub", {im_wh, float_value_1})->output(0); - auto pred_box_xymin_xymax = helper_->Split(pred_box, {2, 2}, 2); - pred_box_xymin_xymax[0] = - helper_->MakeNode("Relu", {pred_box_xymin_xymax[0]})->output(0); - pred_box_xymin_xymax[1] = - helper_->MakeNode("Min", {pred_box_xymin_xymax[1], im_wh})->output(0); - auto out = helper_->Concat(pred_box_xymin_xymax, 2); - helper_->AutoCast(out, boxes_info[0].name, x_info[0].dtype, - boxes_info[0].dtype); - } - - auto class_num = - helper_->Constant({1}, ONNX_NAMESPACE::TensorProto::INT64, class_num_); - auto score_out_shape = - helper_->Concat({nchw[0], value_neg_1, class_num}, int64_t(0)); - auto score_out = - helper_->MakeNode("Reshape", {pred_score, score_out_shape})->output(0); - helper_->AutoCast(score_out, scores_info[0].name, x_info[0].dtype, - scores_info[0].dtype); -} -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/detection/yolo_box.h b/paddle2onnx/mapper/detection/yolo_box.h deleted file mode 100644 index 1c4055f5e26..00000000000 --- a/paddle2onnx/mapper/detection/yolo_box.h +++ /dev/null @@ -1,53 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class YoloBoxMapper : public Mapper { - public: - YoloBoxMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - MarkAsExperimentalOp(); - GetAttr("clip_bbox", &clip_bbox_); - GetAttr("iou_aware", &iou_aware_); - GetAttr("conf_thresh", &conf_thresh_); - GetAttr("iou_aware_factor", &iou_aware_factor_); - GetAttr("class_num", &class_num_); - GetAttr("downsample_ratio", &downsample_ratio_); - GetAttr("scale_x_y", &scale_x_y_); - GetAttr("anchors", &anchors_); - } - - int32_t GetMinOpset(bool verbose = false); - void Opset11(); - - private: - bool clip_bbox_; - bool iou_aware_; - float conf_thresh_; - float iou_aware_factor_; - float scale_x_y_; - int64_t class_num_; - int64_t downsample_ratio_; - std::vector anchors_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/elementwise.cc b/paddle2onnx/mapper/elementwise.cc deleted file mode 100755 index e766d7a3371..00000000000 --- a/paddle2onnx/mapper/elementwise.cc +++ /dev/null @@ -1,187 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. -#include "paddle2onnx/mapper/elementwise.h" - -namespace paddle2onnx { - -REGISTER_MAPPER(elementwise_add, ElementwiseMapper) -REGISTER_MAPPER(elementwise_sub, ElementwiseMapper) -REGISTER_MAPPER(elementwise_div, ElementwiseMapper) -REGISTER_MAPPER(elementwise_mul, ElementwiseMapper) -REGISTER_MAPPER(elementwise_min, ElementwiseMapper) -REGISTER_MAPPER(elementwise_max, ElementwiseMapper) -REGISTER_MAPPER(elementwise_pow, ElementwiseMapper) -REGISTER_MAPPER(elementwise_mod, ElementWiseModMapper) -REGISTER_MAPPER(elementwise_floordiv, ElementWiseFloordivMapper) - -int32_t ElementwiseMapper::GetMinOpset(bool verbose) { - if (OpType() == "elementwise_min" || OpType() == "elementwise_max") { - Logger(verbose, 8) << RequireOpset(8) << std::endl; - return 8; - } - return 7; -} - -void ElementwiseMapper::Opset7() { - auto input_x_info = GetInput("X"); - auto input_y_info = GetInput("Y"); - auto output_info = GetOutput("Out"); - auto iter = op_mapper_.find(OpType()); - Assert(op_mapper_.end() != iter, - "Cannot find " + OpType() + " in elementwise op_mapper."); - - auto x_name = input_x_info[0].name; - auto y_name = input_y_info[0].name; - if (input_x_info[0].dtype == P2ODataType::BOOL && - input_y_info[0].dtype == P2ODataType::BOOL) { - x_name = - helper_->AutoCast(x_name, input_x_info[0].dtype, P2ODataType::INT32); - y_name = - helper_->AutoCast(y_name, input_y_info[0].dtype, P2ODataType::INT32); - } - - std::string output_name; - if (axis_ == -1 || axis_ == (input_x_info[0].Rank() - 1) || - input_x_info[0].Rank() == input_y_info[0].Rank()) { - output_name = helper_->MakeNode(iter->second, {x_name, y_name})->output(0); - } else { - std::vector broadcast_shape(input_x_info[0].Rank(), 1); - for (int i = axis_; i < axis_ + input_y_info[0].Rank(); ++i) { - broadcast_shape[i] = input_y_info[0].shape[i - axis_]; - } - std::string broadcast_shape_node = - helper_->Constant(GetOnnxDtype(P2ODataType::INT64), broadcast_shape); - auto y_node = helper_->MakeNode("Reshape", {y_name, broadcast_shape_node}); - output_name = - helper_->MakeNode(iter->second, {x_name, y_node->output(0)})->output(0); - } - - if (input_x_info[0].dtype == P2ODataType::BOOL && - input_y_info[0].dtype == P2ODataType::BOOL) { - helper_->AutoCast(output_name, output_info[0].name, P2ODataType::INT32, - P2ODataType::BOOL); - } else { - helper_->MakeNode("Identity", {output_name}, {output_info[0].name}); - } -} - -void ElementWiseModMapper::Opset10() { - auto input_x_info = GetInput("X"); - auto input_y_info = GetInput("Y"); - auto output_info = GetOutput("Out"); - int64_t fmod = 0; - if (input_y_info[0].dtype == P2ODataType::INT32 || - input_y_info[0].dtype == P2ODataType::INT64) { - if (this->deploy_backend == "tensorrt") { - auto x = helper_->AutoCast(input_x_info[0].name, input_x_info[0].dtype, - input_y_info[0].dtype); - auto times = - helper_->MakeNode("Div", {input_x_info[0].name, input_y_info[0].name}) - ->output(0); - auto result = - helper_->MakeNode("Mul", {input_y_info[0].name, times})->output(0); - helper_->MakeNode("Sub", {input_x_info[0].name, result}, - {output_info[0].name}); - return; - } - auto mod_node = - helper_->MakeNode("Mod", {input_x_info[0].name, input_y_info[0].name}, - {output_info[0].name}); - AddAttribute(mod_node, "fmod", fmod); - return; - } - - fmod = 1; - - auto abs_x_node = helper_->MakeNode("Abs", {input_x_info[0].name}); - auto abs_y_node = helper_->MakeNode("Abs", {input_y_info[0].name}); - - auto dtype = input_y_info[0].dtype; - - std::string zero_node = helper_->Constant({}, GetOnnxDtype(dtype), 0.0); - - auto mod_node = - helper_->MakeNode("Mod", {abs_x_node->output(0), abs_y_node->output(0)}); - AddAttribute(mod_node, "fmod", fmod); - - auto neg_node = helper_->MakeNode("Neg", {mod_node->output(0)}); - - auto less_node = helper_->MakeNode("Less", {input_x_info[0].name, zero_node}); - - std::string condition_node = - helper_->AutoCast(less_node->output(0), dtype, P2ODataType::BOOL); - - auto mod_res_node = helper_->MakeNode( - "Where", {condition_node, neg_node->output(0), mod_node->output(0)}); - - auto mod_y_add_node = - helper_->MakeNode("Add", {mod_res_node->output(0), input_y_info[0].name}); - - auto mod_y_mul_node = - helper_->MakeNode("Mul", {mod_res_node->output(0), input_y_info[0].name}); - - auto mod_y_mul_less_node = - helper_->MakeNode("Less", {mod_y_mul_node->output(0), zero_node}); - - std::string mod_y_mul_condition_node = helper_->AutoCast( - mod_y_mul_less_node->output(0), dtype, P2ODataType::BOOL); - - helper_->MakeNode("Where", - {mod_y_mul_condition_node, mod_y_add_node->output(0), - mod_res_node->output(0)}, - {output_info[0].name}); -} - -void ElementWiseFloordivMapper::Opset7() { - auto input_x_info = GetInput("X"); - auto input_y_info = GetInput("Y"); - auto output_info = GetOutput("Out"); - - bool is_int = false; - if (input_x_info[0].dtype <= 3 || input_x_info[0].dtype == 20 || - input_y_info[0].dtype <= 3 || input_y_info[0].dtype == 20) { - is_int = true; - } - if (axis_ == -1 || axis_ == input_x_info[0].Rank() - 1 || - input_x_info[0].Rank() == input_y_info[0].Rank()) { - if (is_int) { - helper_->MakeNode("Div", {input_x_info[0].name, input_y_info[0].name}, - {output_info[0].name}); - } else { - auto div_node = helper_->MakeNode( - "Div", {input_x_info[0].name, input_y_info[0].name}); - helper_->MakeNode("Floor", {div_node->output(0)}, {output_info[0].name}); - } - } else { - std::vector broadcast_shape; - broadcast_shape.resize(axis_ + input_x_info[0].Rank(), 1); - for (auto i = 0; i < input_y_info[0].Rank(); ++i) { - broadcast_shape[axis_ + i] = input_y_info[0].shape[i]; - } - std::string broadcast_shape_node = - helper_->Constant(GetOnnxDtype(P2ODataType::INT64), broadcast_shape); - auto y_node = helper_->MakeNode( - "Reshape", {input_y_info[0].name, broadcast_shape_node}); - if (is_int) { - helper_->MakeNode("Div", {input_x_info[0].name, y_node->output(0)}, - {output_info[0].name}); - } else { - auto div_node = - helper_->MakeNode("Div", {input_x_info[0].name, y_node->output(0)}); - helper_->MakeNode("Floor", {div_node->output(0)}, {output_info[0].name}); - } - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/elementwise.h b/paddle2onnx/mapper/elementwise.h deleted file mode 100644 index 4a6ec28f912..00000000000 --- a/paddle2onnx/mapper/elementwise.h +++ /dev/null @@ -1,75 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. -#pragma once -#include -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class ElementwiseMapper : public Mapper { - public: - ElementwiseMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("axis", &axis_); - - op_mapper_["elementwise_add"] = "Add"; - op_mapper_["elementwise_sub"] = "Sub"; - op_mapper_["elementwise_div"] = "Div"; - op_mapper_["elementwise_mul"] = "Mul"; - op_mapper_["elementwise_min"] = "Min"; - op_mapper_["elementwise_max"] = "Max"; - op_mapper_["elementwise_pow"] = "Pow"; - } - - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - std::map op_mapper_; - int64_t axis_; -}; - -class ElementWiseModMapper : public Mapper { - public: - ElementWiseModMapper(const PaddleParser& p, OnnxHelper* helper, - int64_t block_id, int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - - int32_t GetMinOpset(bool verbose = false) { - Logger(verbose, 10) << RequireOpset(10) << std::endl; - return 10; - } - - void Opset10(); -}; - -class ElementWiseFloordivMapper : public Mapper { - public: - ElementWiseFloordivMapper(const PaddleParser& p, OnnxHelper* helper, - int64_t block_id, int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("axis", &axis_); - } - - void Opset7(); - - private: - int64_t axis_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/exporter.cc b/paddle2onnx/mapper/exporter.cc deleted file mode 100644 index 595d4d2267b..00000000000 --- a/paddle2onnx/mapper/exporter.cc +++ /dev/null @@ -1,631 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/exporter.h" - -#include -#include - -#include - -#include "onnxoptimizer/optimize.h" -#include "paddle2onnx/optimizer/convert_fp32_to_fp16.h" -#include "paddle2onnx/optimizer/eliminate_non_transpose.h" -#include "paddle2onnx/optimizer/fuse_constant_cast.h" -#include "paddle2onnx/optimizer/fuse_constant_reshape.h" -#include "paddle2onnx/optimizer/fuse_constant_unsqueeze.h" -#include "paddle2onnx/optimizer/fuse_paddle_conv_bias.h" -#include "paddle2onnx/optimizer/fuse_unsqueeze_conv2d_squeeze.h" - -namespace paddle2onnx { -MapperHelper* MapperHelper::helper = nullptr; - -void ModelExporter::ExportParameters( - const std::map& params, bool use_initializer) { - for (auto& item : params) { - // TODO(jiangjiajun) I'm not handling use_initializer now, but some day I - // will - auto node = MakeConstant(item.first, item.second); - parameters.push_back(std::move(node)); - } -} - -void ModelExporter::UpdateParameters( - const std::map& params) { - for (auto& item : params) { - auto node = MakeConstant(item.first, item.second); - bool updated = false; - for (int i = 0; i < parameters.size(); ++i) { - auto old_node = parameters[i]; - if (old_node->output(0) == item.first) { - parameters.erase(parameters.begin() + i); - parameters.push_back(std::move(node)); - updated = true; - break; - } - } - if (!updated) { - parameters.push_back(std::move(node)); - } - } -} - -void ModelExporter::ExportInputOutputs( - const std::vector& input_infos, - const std::vector& output_infos) { - for (auto& item : input_infos) { - auto value_info = MakeValueInfo(item); - inputs.push_back(std::move(value_info)); - } - for (auto& item : output_infos) { - auto value_info = MakeValueInfo(item); - outputs.push_back(std::move(value_info)); - } -} - -void ModelExporter::CovertCustomOps(const PaddleParser& parser, - OnnxHelper* helper, int64_t block_id, - int64_t op_id) { - auto op = parser.GetOpDesc(block_id, op_id); - std::vector input_strs; - for (auto i_index = 0; i_index < op.inputs_size(); i_index++) { - auto input = op.inputs(i_index); - std::string parameter = input.parameter(); - if (parser.OpHasInput(block_id, op_id, parameter)) { - auto input_info = parser.GetOpInput(block_id, op_id, parameter); - for (auto input : input_info) { - input_strs.push_back(input.name); - helper->MakeValueInfo(input.name, input.dtype, input.shape); - } - } - } - std::vector output_strs; - for (auto o_index = 0; o_index < op.outputs_size(); o_index++) { - auto output = op.outputs(o_index); - std::string parameter = output.parameter(); - if (parser.OpHasOutput(block_id, op_id, parameter)) { - auto output_info = parser.GetOpOutput(block_id, op_id, parameter); - for (auto output : output_info) { - output_strs.push_back(output.name); - helper->MakeValueInfo(output.name, output.dtype, output.shape); - } - } - } - auto node = helper->MakeNode(custom_ops[op.type()], input_strs, output_strs); - node->set_domain("Paddle"); - for (auto attr_index = 0; attr_index < op.attrs_size(); attr_index++) { - auto attr = op.attrs(attr_index); - std::string attr_name = attr.name(); - if (attr_name == "op_callstack") { - continue; - } - if (attr.has_i() || attr.has_l()) { - int64_t val; - parser.GetOpAttr(op, attr_name, &val); - AddAttribute(node, attr_name, val); - } else if (attr.has_f()) { - float val; - parser.GetOpAttr(op, attr_name, &val); - AddAttribute(node, attr_name, val); - } else if (attr.has_b()) { - bool val; - parser.GetOpAttr(op, attr_name, &val); - AddAttribute(node, attr_name, static_cast(val)); - } else if (attr.has_s()) { - std::string val; - parser.GetOpAttr(op, attr_name, &val); - AddAttribute(node, attr_name, val); - } else if (attr.ints_size() > 0 || attr.longs_size() > 0) { - std::vector vec; - parser.GetOpAttr(op, attr_name, &vec); - AddAttribute(node, attr_name, vec); - } else if (attr.floats_size() > 0) { - std::vector vec; - parser.GetOpAttr(op, attr_name, &vec); - AddAttribute(node, attr_name, vec); - } else if (attr.float64s_size() > 0) { - std::vector vec; - parser.GetOpAttr(op, attr_name, &vec); - std::vector fp32_vec; - for (auto val : vec) { - fp32_vec.push_back(static_cast(val)); - } - AddAttribute(node, attr_name, fp32_vec); - } - } - P2OLogger(true) << op.type() << " is exported as custom operator: " - << custom_ops[op.type()] << std::endl; -} - -void ModelExporter::ExportOp(const PaddleParser& parser, OnnxHelper* helper, - int32_t opset_version, int64_t block_id, - int64_t op_id, bool verbose) { - _current_exported_num += 1; - auto op = parser.GetOpDesc(block_id, op_id); -#ifdef PADDLE2ONNX_DEBUG - P2OLogger(true) << "---Converting operator: " << op.type() << " ---" - << std::endl; -#endif - if (op.type() == "while") { - return ExportLoop(parser, helper, opset_version, block_id, op_id, verbose); - } - - if (MapperHelper::Get()->IsRegistered(op.type())) { - auto mapper = MapperHelper::Get()->CreateMapper(op.type(), parser, helper, - block_id, op_id); - mapper->deploy_backend = _deploy_backend; -#ifdef PADDLE2ONNX_DEBUG - P2OLogger(true) << "Mapper Name: " << mapper->Name() << std::endl; -#endif - // Some operators will export as custom operator - auto iter = custom_ops.find(op.type()); - if (iter != custom_ops.end()) { - mapper->export_as_custom_op = true; - mapper->custom_op_name = iter->second; - } - mapper->Run(); - delete mapper; - } else if (custom_ops.find(op.type()) != custom_ops.end()) { - CovertCustomOps(parser, helper, block_id, op_id); - } - -#ifdef PADDLE2ONNX_DEBUG - P2OLogger(true) << "---Converting operator: " << op.type() << " done---" - << std::endl; -#endif -} - -void ModelExporter::ProcessGraphDumplicateNames( - std::vector>* parameters, - std::vector>* inputs, - std::vector>* outputs, - std::vector>* nodes, - std::map* quantize_info) { - // process dumplicate tensor names - std::map renamer; - std::set tensor_names; - for (auto& item : *parameters) { - for (size_t i = 0; i < item->output_size(); ++i) { - if (tensor_names.find(item->output(i)) != tensor_names.end()) { - Assert(false, "There's dumplicate names in exported parameters."); - } - tensor_names.insert(item->output(i)); - } - } - for (auto& item : *inputs) { - if (tensor_names.find(item->name()) != tensor_names.end()) { - Assert(false, "There's dumplicate names:" + item->name() + - " in exported parameters and inputs."); - } - tensor_names.insert(item->name()); - } - for (auto& item : *nodes) { - // update node inputs - for (size_t i = 0; i < item->input_size(); ++i) { - if (renamer.find(item->input(i)) != renamer.end()) { - auto updated_name = renamer[item->input(i)]; - while (renamer.find(updated_name) != renamer.end()) { - updated_name = renamer[updated_name]; - } - *(item->mutable_input(i)) = updated_name; - } - } - // if there's dumplicate name - // will generate new name and replace it - for (size_t i = 0; i < item->output_size(); ++i) { - if (tensor_names.find(item->output(i)) != tensor_names.end()) { - std::string renamed_tensor_name = item->output(i); - while (renamer.find(renamed_tensor_name) != renamer.end()) { - renamed_tensor_name = renamer[renamed_tensor_name]; - } - auto new_tensor_name = - MapperHelper::Get()->GenName(renamed_tensor_name); - P2OLogger() << "Find dumplicate output name '" << renamed_tensor_name - << "', it will rename to '" << new_tensor_name << "'." - << std::endl; - if (quantize_info && - quantize_info->find(renamed_tensor_name) != quantize_info->end()) { - (*quantize_info)[new_tensor_name] = - (*quantize_info)[renamed_tensor_name]; - } - *(item->mutable_output(i)) = new_tensor_name; - renamer[renamed_tensor_name] = new_tensor_name; - } - tensor_names.insert(item->output(i)); - } - } - - for (auto& item : *outputs) { - if (renamer.find(item->name()) != renamer.end()) { - auto updated_name = renamer[item->name()]; - while (renamer.find(updated_name) != renamer.end()) { - updated_name = renamer[updated_name]; - } - item->set_name(updated_name); - } - } -} - -void ModelExporter::SaveExternalData(::paddle2onnx::GraphProto* graph, - const std::string& external_file_path, - bool* save_external) { - P2OLogger() << "The exported ONNX model is bigger than 2G, external data " - "will save to file: " - << external_file_path << std::endl; - std::string file_name = GetFilenameFromPath(external_file_path); - if (save_external) { - *save_external = true; - } - std::fstream f(external_file_path, std::ios::out); - Assert(f.is_open(), "Failed to open: " + external_file_path + - " file to save external data"); - for (auto index = 0; index < graph->node_size(); index++) { - auto node = graph->mutable_node(index); - if (node->op_type() != "Constant") { - continue; - } - for (auto i = 0; i < node->attribute_size(); i++) { - auto attr = node->mutable_attribute(i); - if (attr->name() != "value") { - continue; - } - auto tensor = attr->mutable_t(); - - if (tensor->raw_data().size() <= 128) { - continue; - } - - tensor->set_data_location(TensorProto::EXTERNAL); - auto external_data = tensor->add_external_data(); - external_data->set_key("location"); - external_data->set_value(file_name); - - external_data = tensor->add_external_data(); - external_data->set_key("offset"); - f.seekg(0, std::ios::end); - int64_t offset = f.tellg(); - external_data->set_value(std::to_string(offset)); - auto raw_data = tensor->raw_data(); - f << raw_data; - external_data = tensor->add_external_data(); - external_data->set_key("length"); - int64_t raw_datas_size = raw_data.size(); - external_data->set_value(std::to_string(raw_datas_size)); - tensor->clear_raw_data(); - } - } - f.close(); -} -void ModelExporter::ONNXChecker(const ONNX_NAMESPACE::ModelProto& model, - const bool& verbose) { - // TODO(jiangjiajun) - // If we need to integrate with framework - // this check will return a information - // to let framework know the conversion is - // pass or fail - try { - // ONNX_NAMESPACE::checker::check_model(*(model.get())); - ONNX_NAMESPACE::checker::check_model(model); - } catch (const std::exception& e) { - P2OLogger(verbose) << "The exported ONNX model is invalid." << std::endl; - P2OLogger(verbose) << "Model checker error log: " << e.what() << std::endl; - } - P2OLogger(verbose) << "PaddlePaddle model is exported as ONNX format now." - << std::endl; -} - -std::string ModelExporter::Run( - const PaddleParser& parser, int opset_version, bool auto_upgrade_opset, - bool verbose, bool enable_onnx_checker, bool enable_experimental_op, - bool enable_optimize, const std::string& deploy_backend, - std::string* calibration_cache, const std::string& external_file, - bool* save_external, bool export_fp16_model, - std::vector disable_fp16_op_types) { - _deploy_backend = deploy_backend; - _helper.SetOpsetVersion(opset_version); - _total_ops_num = 0; - _current_exported_num = 0; - for (auto i = 0; i < parser.NumOfBlocks(); ++i) { - _total_ops_num += parser.NumOfOps(i); - } - _helper.nodes.reserve(_total_ops_num * 3); - Assert(opset_version <= MAX_ONNX_OPSET_VERSION && opset_version >= 7, - "Paddle2ONNX now only support opset version in range of [7, " + - std::to_string(MAX_ONNX_OPSET_VERSION) + "]."); - _helper.Clear(); - inputs.clear(); - outputs.clear(); - parameters.clear(); - - // clear name_counter - // this use to generate unique name - // for intermdiate - // while converting all the op - MapperHelper::Get()->ClearNameCounter(); - - std::set unsupported_ops; - if (!CheckIfOpSupported(parser, &unsupported_ops, enable_experimental_op)) { - auto logger = P2OLogger(); - logger << "Oops, there are some operators not supported yet, including "; - for (auto& item : unsupported_ops) { - logger << item << ","; - } - logger << std::endl; - Assert(1 == 0, - "Due to the unsupported operators, the conversion is aborted."); - } - - int32_t min_opset = GetMinOpset(parser, verbose); - if (min_opset < 0) { - Assert(false, - "Model exporting failed, you can report this problem to " - "https://github.com/PaddlePaddle/Paddle2ONNX.git."); - } - if (!auto_upgrade_opset) { - if (min_opset > opset_version) { - P2OLogger() << "This PaddlePaddle model is not able to export to ONNX " - "with opset_version=" - << opset_version << ", please set the opset_version to " - << min_opset << " or higher for successfully conversion." - << std::endl; - Assert(false, - "Due to opset version, the model exporting is aborted, please set " - "a higher opset_version or set auto_upgrade_opset=true."); - } - } else { - if (min_opset > opset_version) { - P2OLogger() << "Opset version will change to " << min_opset << " from " - << opset_version << std::endl; - opset_version = min_opset; - } - } - _helper.SetOpsetVersion(opset_version); - P2OLogger(verbose) << "Use opset_version = " << _helper.GetOpsetVersion() - << " for ONNX export." << std::endl; - ExportParameters(parser.params); - ExportInputOutputs(parser.inputs, parser.outputs); - - // Only convert blocks 0 now - // because control flow is not supported yet - for (auto i = 0; i < parser.NumOfOps(0); ++i) { - auto op = parser.GetOpDesc(0, i); - if (op.type() == "feed") { - continue; - } else if (op.type() == "fetch") { - continue; - } - ExportOp(parser, &_helper, opset_version, 0, i, verbose); - } - - // construct a onnx model proto - auto model = std::make_shared(); - // TODO(jiangjiajun) ir version is related to onnx version - model->set_ir_version(ONNX_NAMESPACE::IR_VERSION); - auto graph = model->mutable_graph(); - graph->set_name("Model from PaddlePaddle."); - auto opset_id = model->add_opset_import(); - opset_id->set_domain(""); - opset_id->set_version(opset_version); - if (custom_ops.size()) { - auto opset_paddle_id = model->add_opset_import(); - opset_paddle_id->set_domain("Paddle"); - opset_paddle_id->set_version(1); - } - - ProcessGraphDumplicateNames(¶meters, &inputs, &outputs, &_helper.nodes, - &_helper.quantize_info); - if (parser.is_quantized_model) { - quantize_model_processer.ProcessQuantizeModel( - ¶meters, &inputs, &outputs, &_helper.nodes, &_helper, - deploy_backend, parser, calibration_cache); - // Update int8 weights in quantized OP to float32 - UpdateParameters(_helper.updated_params); - } - - for (auto& item : parameters) { - *(graph->add_node()) = *(item.get()); - } - for (auto& item : inputs) { - *(graph->add_input()) = *(item.get()); - } - for (auto& item : _helper.nodes) { - *(graph->add_node()) = (*item.get()); - } - for (auto& item : outputs) { - *(graph->add_output()) = (*item.get()); - } - for (auto& item : _helper.value_infos) { - *(graph->add_value_info()) = (*item.get()); - } - - ONNX_NAMESPACE::ModelProto onnx_model; - std::string out; - if (enable_optimize) { - onnx_model = Optimize(*(model.get())); - } else { - onnx_model = *model.get(); - } - - // convert fp32 model to fp16 - if (export_fp16_model) { - P2OLogger(verbose) << "Convert FP32 ONNX model to FP16." << std::endl; - ConvertFp32ToFp16 convert; - convert.SetCustomOps(custom_ops); - convert.AddDisabledOpTypes(disable_fp16_op_types); - convert.Convert(&onnx_model); - } - - // save external data file for big model - std::string external_data_file; - if (onnx_model.ByteSizeLong() > INT_MAX) { - if (external_file.empty()) { - external_data_file = "external_data"; - } else { - external_data_file = external_file; - } - } - if (external_data_file.size()) { - SaveExternalData(onnx_model.mutable_graph(), external_data_file, - save_external); - } - // check model - if (enable_onnx_checker) { - ONNXChecker(onnx_model, verbose); - } - - if (!onnx_model.SerializeToString(&out)) { - P2OLogger(verbose) - << "Error happenedd while optimizing the exported ONNX model." - << std::endl; - return ""; - } - return out; -} - -bool ModelExporter::CheckIfOpSupported(const PaddleParser& parser, - std::set* unsupported_ops, - bool enable_experimental_op) { - unsupported_ops->clear(); - for (auto i = 0; i < parser.NumOfBlocks(); ++i) { - for (auto j = 0; j < parser.NumOfOps(i); ++j) { - auto op = parser.GetOpDesc(i, j); - if (op.type() == "feed" || op.type() == "fetch") { - continue; - } - if (op.type() == "while" && enable_experimental_op) { - if (!IsLoopSupported(parser, i, j)) { - unsupported_ops->insert("while"); - } - continue; - } - if (custom_ops.find(op.type()) != custom_ops.end()) { - continue; - } - if (!MapperHelper::Get()->IsRegistered(op.type())) { - unsupported_ops->insert(op.type()); - } else if (!enable_experimental_op) { - auto mapper = MapperHelper::Get()->CreateMapper(op.type(), parser, - &_helper, i, j); - if (mapper->IsExperimentalOp()) { - unsupported_ops->insert(op.type()); - } - delete mapper; - } - } - } - return (unsupported_ops->size() == 0); -} - -int32_t ModelExporter::GetMinOpset(const PaddleParser& parser, bool verbose) { - int32_t opset_version = _helper.GetOpsetVersion(); - int32_t max_opset = 7; - bool exportable = true; - // Record the number of ops that need to be converted - int converted_op_num = 0; - std::set verbose_log; - for (auto i = 0; i < parser.NumOfBlocks(); ++i) { - for (auto j = 0; j < parser.NumOfOps(i); ++j) { - auto op = parser.GetOpDesc(i, j); - if (custom_ops.find(op.type()) != custom_ops.end()) { - continue; - } - if (op.type() == "feed" || op.type() == "fetch") { - continue; - } - converted_op_num += 1; - int current_min_opset = 7; - if (op.type() == "while") { - P2OLogger() << "Detected there's control flow 'while' op in your " - "model, this requires the minimal opset version of 13." - << std::endl; - current_min_opset = 13; - } else { - auto mapper = MapperHelper::Get()->CreateMapper(op.type(), parser, - &_helper, i, j); - auto iter = custom_ops.find(op.type()); - if (iter != custom_ops.end()) { - mapper->export_as_custom_op = true; - } - current_min_opset = mapper->GetMinOpset(verbose); - delete mapper; - } - if (current_min_opset < 0) { - exportable = false; - P2OLogger(verbose) << "Due to the operator: " << op.type() - << ", this model cannot be exported to ONNX." - << std::endl; - } else if (current_min_opset > max_opset) { - max_opset = current_min_opset; - if (verbose && current_min_opset > opset_version) { - verbose_log.insert("Due to the operator: " + op.type() + - ", requires opset_version >= " + - std::to_string(current_min_opset) + "."); - } - } - } - } - if (verbose) { - for (auto iter = verbose_log.begin(); iter != verbose_log.end(); ++iter) { - P2OLogger() << *iter << std::endl; - } - } - - // Here we put some checks to make sure - // paddle2onnx could compatible with - // other version of onnx - int32_t max_support_opset = MAX_ONNX_OPSET_VERSION; - if (exportable && (max_opset > MAX_ONNX_OPSET_VERSION)) { - exportable = false; - P2OLogger() << "[ERROR] The compiled ONNX version only supports opset 7~" - << MAX_ONNX_OPSET_VERSION - << ", but now this model need as least opset " << max_opset - << ", please compile with higher version of ONNX." << std::endl; - } - if (exportable) { - return max_opset; - } - - return -1; -} - -ONNX_NAMESPACE::ModelProto ModelExporter::Optimize( - const ONNX_NAMESPACE::ModelProto& model) { - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - std::vector passes = {"eliminate_identity", - "eliminate_deadend", - "eliminate_deadend", - "fuse_constant_reshape", - "fuse_constant_unsqueeze", - "fuse_paddle_conv_bias", - "fuse_consecutive_transposes", - "eliminate_non_transpose", - "fuse_matmul_add_bias_into_gemm", - "eliminate_identity", - "eliminate_deadend", - "eliminate_unused_initializer"}; - return ONNX_NAMESPACE::optimization::Optimize(model, passes); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/exporter.h b/paddle2onnx/mapper/exporter.h deleted file mode 100644 index 795198ee5aa..00000000000 --- a/paddle2onnx/mapper/exporter.h +++ /dev/null @@ -1,123 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include - -#include -#include - -#include "paddle2onnx/mapper/mapper.h" -#include "paddle2onnx/mapper/quantize_helper.h" -#include "paddle2onnx/parser/parser.h" - -#ifdef _MSC_VER -#define PATH_SEP "\\" -#else -#define PATH_SEP "/" -#endif - -inline std::string GetFilenameFromPath(const std::string& path) { - auto pos = path.find_last_of(PATH_SEP); - if (pos == std::string::npos) { - return path; - } - return path.substr(pos + 1); -} - -namespace paddle2onnx { - -struct ModelExporter { - private: - std::vector> parameters; - std::vector> inputs; - std::vector> outputs; - // The _deploy_backend will pass to Mapper to influence the conversion - std::string _deploy_backend = "onnxruntime"; - OnnxHelper _helper; - int32_t _total_ops_num = 0; - int32_t _current_exported_num = 0; - - void ExportParameters(const std::map& params, - bool use_initializer = false); - - // Update constant node in parameters. When process quantize model, the weight - // dtype may be int8, it should be convet to float32 and use this function to - // update converted params. - void UpdateParameters(const std::map& params); - void ExportInputOutputs(const std::vector& input_infos, - const std::vector& output_infos); - void ExportOp(const PaddleParser& parser, OnnxHelper* helper, - int32_t opset_version, int64_t block_id, int64_t op_id, - bool verbose); - bool IsLoopSupported(const PaddleParser& parser, const int64_t& block_id, - const int64_t& op_id); - void ExportLoop(const PaddleParser& parser, OnnxHelper* helper, - int32_t opset_version, int64_t block_id, int64_t op_id, - bool verbose); - void CovertCustomOps(const PaddleParser& parser, OnnxHelper* helper, - int64_t block_id, int64_t op_id); - ONNX_NAMESPACE::ModelProto Optimize(const ONNX_NAMESPACE::ModelProto& model); - - public: - // custom operators for export - // - std::map custom_ops; - - QuantizeModelProcessor quantize_model_processer; - // Get a proper opset version in range of [7, 16] - // Also will check the model is convertable, this will include 2 parts - // 1. is the op convert function implemented - // 2. is the op convertable(some cases may not be able to convert) - // If the model is not convertable, return -1 - int32_t GetMinOpset(const PaddleParser& parser, bool verbose = false); - - // // Remove isolated nodes in onnx model - // void RemoveIsolatedNodes( - // std::vector>* parameters, - // std::vector>* inputs, - // std::vector>* outputs, - // std::vector>* nodes); - // Process dumplicate tensor names in paddle model - void ProcessGraphDumplicateNames( - std::vector>* parameters, - std::vector>* inputs, - std::vector>* outputs, - std::vector>* nodes, - std::map* quantize_info = nullptr); - - bool CheckIfOpSupported(const PaddleParser& parser, - std::set* unsupported_ops, - bool enable_experimental_op); - - void SaveExternalData(::paddle2onnx::GraphProto* graph, - const std::string& external_file_path, - bool* save_external = nullptr); - - void ONNXChecker(const ONNX_NAMESPACE::ModelProto& model, - const bool& verbose); - - std::string Run(const PaddleParser& parser, int opset_version = 9, - bool auto_upgrade_opset = true, bool verbose = false, - bool enable_onnx_checker = true, - bool enable_experimental_op = false, - bool enable_optimize = true, - const std::string& deploy_backend = "onnxruntime", - std::string* calibration_cache = nullptr, - const std::string& external_file = "", - bool* save_external = nullptr, bool export_fp16_model = false, - std::vector disable_fp16_op_types = {}); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/loop.cc b/paddle2onnx/mapper/loop.cc deleted file mode 100644 index 00d72b32334..00000000000 --- a/paddle2onnx/mapper/loop.cc +++ /dev/null @@ -1,195 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/exporter.h" - -namespace paddle2onnx { - -bool ModelExporter::IsLoopSupported(const PaddleParser& parser, - const int64_t& block_id, - const int64_t& op_id) { - auto x_info = parser.GetOpInput(block_id, op_id, "X"); - auto out_info = parser.GetOpOutput(block_id, op_id, "Out"); - auto cond_info = parser.GetOpInput(block_id, op_id, "Condition"); - std::set input_names; - for (size_t i = 0; i < x_info.size(); ++i) { - input_names.insert(x_info[i].name); - } - input_names.insert(cond_info[0].name); - - for (size_t i = 0; i < out_info.size(); ++i) { - auto iter = input_names.find(out_info[i].name); - if (iter == input_names.end()) { - P2OLogger() << "Cannot find output:" << out_info[i].name << " in input tensors while converting operator 'while', Paddle2ONNX doesn't support this situation now." << std::endl; - return false; - } - } - for (size_t i = 0; i < x_info.size(); ++i) { - if (x_info[i].is_tensor_array) { - P2OLogger() << "LodTensorArray is not supported." << std::endl; - return false; - } - } - return true; -} - -void ModelExporter::ExportLoop(const PaddleParser& parser, OnnxHelper* helper, - int32_t opset_version, int64_t block_id, - int64_t op_id, bool verbose) { - auto op = parser.GetOpDesc(block_id, op_id); - int32_t sub_block_idx = -1; - for (size_t i = 0; i < op.attrs_size(); ++i) { - if (op.attrs(i).name() == "sub_block") { - sub_block_idx = op.attrs(i).block_idx(); - break; - } - } - Assert(sub_block_idx > 0, "Cannot find sub_block in while operator."); - auto x_info = parser.GetOpInput(block_id, op_id, "X"); - auto cond_info = parser.GetOpInput(block_id, op_id, "Condition"); - - std::vector> inputs; - std::vector> outputs; - - // make loop iter - auto iter_name = MapperHelper::Get()->GenName("loop.iter"); - TensorInfo iter_info(iter_name, std::vector(1, 1), - P2ODataType::INT64); - inputs.push_back(std::move(MakeValueInfo(iter_info))); - - std::set input_names; - // make cond - inputs.push_back(std::move(MakeValueInfo(cond_info[0]))); - input_names.insert(cond_info[0].name); - // other inputs - outputs.push_back(std::move(std::move(MakeValueInfo(cond_info[0])))); - for (size_t i = 0; i < x_info.size(); ++i) { - if (x_info[i].is_tensor_array) { - continue; - } - if (input_names.find(x_info[i].name) != input_names.end()) { - continue; - } - input_names.insert(x_info[i].name); - inputs.push_back(std::move(MakeValueInfo(x_info[i]))); - outputs.push_back(std::move(MakeValueInfo(x_info[i]))); - } - for (size_t i = 0; i < x_info.size(); ++i) { - if (x_info[i].is_tensor_array) { - if (input_names.find(x_info[i].name) != input_names.end()) { - continue; - } - input_names.insert(x_info[i].name); - outputs.push_back(std::move(MakeValueInfo(x_info[i]))); - } - } - - // make op nodes - OnnxHelper loop_helper; - loop_helper.SetOpsetVersion(opset_version); - - for (auto i = 0; i < parser.NumOfOps(sub_block_idx); ++i) { - auto op = parser.GetOpDesc(sub_block_idx, i); - ExportOp(parser, &loop_helper, opset_version, sub_block_idx, i, verbose); - } - - std::vector> parameters; - ProcessGraphDumplicateNames(¶meters, &inputs, &outputs, - &loop_helper.nodes); - std::map renamer; - for (auto& item : inputs) { - auto name = MapperHelper::Get()->GenName("loop.input"); - renamer[item->name()] = name; - item->set_name(name); - } - for (auto& item : loop_helper.nodes) { - for (size_t i = 0; i < item->input_size(); ++i) { - if (renamer.find(item->input(i)) != renamer.end()) { - auto updated_name = renamer[item->input(i)]; - while (renamer.find(updated_name) != renamer.end()) { - updated_name = renamer[updated_name]; - } - *(item->mutable_input(i)) = updated_name; - } - } - } - for (auto& item : outputs) { - if (renamer.find(item->name()) != renamer.end()) { - auto updated_name = renamer[item->name()]; - while (renamer.find(updated_name) != renamer.end()) { - updated_name = renamer[updated_name]; - } - item->set_name(updated_name); - } - } - - // // construct a onnx model proto - // // consider to optimize the subgraph - // auto model = std::make_shared(); - // model->set_ir_version(ONNX_NAMESPACE::IR_VERSION); - // auto graph = model->mutable_graph(); - // auto graph_name = MapperHelper::Get()->GenName("Model from - // PaddlePaddle(Loop)."); - // graph->set_name(graph_name); - // auto opset_id = model->add_opset_import(); - // opset_id->set_domain(""); - // opset_id->set_version(loop_helper->GetOpsetVersion()); - - auto graph_name = MapperHelper::Get()->GenName("paddle.loop"); - auto graph = std::make_shared(); - graph->set_name(graph_name); - for (auto& item : inputs) { - *(graph->add_input()) = *(item.get()); - } - for (auto& item : loop_helper.nodes) { - *(graph->add_node()) = (*item.get()); - } - for (auto& item : outputs) { - *(graph->add_output()) = (*item.get()); - } - - // fake iter - auto fake_iter = helper->Constant(ONNX_NAMESPACE::TensorProto::INT64, - std::vector(1, 1024)); - std::vector x_names; - x_names.push_back(fake_iter); - x_names.push_back(cond_info[0].name); - std::vector out_names; - for (size_t i = 0; i < x_info.size(); ++i) { - if (x_info[i].is_tensor_array) { - continue; - } - if (std::find(x_names.begin(), x_names.end(), x_info[i].name) != x_names.end()) { - continue; - } - x_names.push_back(x_info[i].name); - out_names.push_back(x_info[i].name); - } - for (size_t i = 0; i < x_info.size(); ++i) { - if (x_info[i].is_tensor_array) { - if (std::find(x_names.begin(), x_names.end(), x_info[i].name) != x_names.end()) { - continue; - } - out_names.push_back(x_info[i].name); - } - } - - auto loop_node = helper->MakeNode("Loop", x_names, out_names); - auto attr = loop_node->add_attribute(); - attr->set_name("body"); - attr->set_type(ONNX_NAMESPACE::AttributeProto::GRAPH); - *(attr->mutable_g()) = *(graph.get()); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/mapper.h b/paddle2onnx/mapper/mapper.h deleted file mode 100755 index 1cb72d59f36..00000000000 --- a/paddle2onnx/mapper/mapper.h +++ /dev/null @@ -1,246 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. -#pragma once -#include - -#include "paddle2onnx/mapper/data_helper.h" -#include "paddle2onnx/mapper/onnx_helper.h" -#include "paddle2onnx/mapper/register_mapper.h" -#include "paddle2onnx/parser/parser.h" - -namespace paddle2onnx { - -class Mapper { - public: - Mapper() {} - Mapper(const PaddleParser& p, OnnxHelper* helper, int32_t block_id, - int32_t op_id, std::string name = {}) - : parser_(&p) { - block_idx_ = block_id; - op_idx_ = op_id; - helper_ = helper; - name_ = name; - } - - // The flag will control if the op is exported as a custom operator - // if export_as_custom_op = true, will exported as description in - // custom_op_info - bool export_as_custom_op = false; - // [exported_op_name, domain] - std::string custom_op_name; - std::string deploy_backend; - - P2OLogger Logger(const bool& verbose, const int32_t& opset_version = 100) { - bool v = verbose; - if (opset_version <= helper_->GetOpsetVersion()) { - v = false; - } - auto& op = parser_->GetOpDesc(block_idx_, op_idx_); - std::string output_name = ""; - if (op.outputs(0).arguments_size() > 0) { - output_name = op.outputs(0).arguments(0); - } - std::string op_type = op.type(); - std::string prefix = "[Paddle2ONNX] [" + op_type + ": " + output_name + "]"; - return P2OLogger(v, prefix); - } - - P2OLogger Error() { - auto& op = parser_->GetOpDesc(block_idx_, op_idx_); - std::string output_name = ""; - if (op.outputs(0).arguments_size() > 0) { - output_name = op.outputs(0).arguments(0); - } - std::string op_type = op.type(); - std::string prefix = - "[ERROR][Paddle2ONNX] [" + op_type + ": " + output_name + "]"; - return P2OLogger(true, prefix); - } - - P2OLogger Warn() { - auto& op = parser_->GetOpDesc(block_idx_, op_idx_); - std::string output_name = ""; - if (op.outputs(0).arguments_size() > 0) { - output_name = op.outputs(0).arguments(0); - } - std::string op_type = op.type(); - std::string prefix = - "[WARN][Paddle2ONNX] [" + op_type + ": " + output_name + "]"; - return P2OLogger(true, prefix); - } - - // Some operators is not implement very well, e.g the output may not be same - // We mark these operators as experimental, these operators requires double - // checking after model exported. - virtual void MarkAsExperimentalOp() { is_experimental_op_ = true; } - virtual bool IsExperimentalOp() const { return is_experimental_op_; } - // the return value in [7, MAX_ONNX_OPSET_VERSION], represent the minimum - // opset_version - // if return value < 0, means the op is not supported. - virtual int32_t GetMinOpset(bool verbose = false) { return 7; } - - virtual bool IsExportAsCustomOp() { return export_as_custom_op; } - - void Run() { - int32_t opset_version = helper_->GetOpsetVersion(); - Assert(opset_version >= 7 && opset_version <= MAX_ONNX_OPSET_VERSION, - "[Paddle2ONNX] Only support opset_version in range of [7, " + - std::to_string(MAX_ONNX_OPSET_VERSION) + "]."); - if (IsExportAsCustomOp()) { - return ExportAsCustomOp(); - } - if (opset_version == 16) { - Opset16(); - } else if (opset_version == 15) { - Opset15(); - } else if (opset_version == 14) { - Opset14(); - } else if (opset_version == 13) { - Opset13(); - } else if (opset_version == 12) { - Opset12(); - } else if (opset_version == 11) { - Opset11(); - } else if (opset_version == 10) { - Opset10(); - } else if (opset_version == 9) { - Opset9(); - } else if (opset_version == 8) { - Opset8(); - } else { - Opset7(); - } - } - - virtual void ExportAsCustomOp() { - Assert(false, - "Operator " + name_ + "doesn't support export as custom operator."); - } - - virtual void Opset16() { Opset15(); } - - virtual void Opset15() { Opset14(); } - - virtual void Opset14() { Opset13(); } - - virtual void Opset13() { Opset12(); } - - virtual void Opset12() { Opset11(); } - - virtual void Opset11() { Opset10(); } - - virtual void Opset10() { Opset9(); } - - virtual void Opset9() { Opset8(); } - - virtual void Opset8() { Opset7(); } - - virtual void Opset7() { - Assert(false, - "This error shouldn't happend, please report to " - "https://github.com/PaddlePaddle/Paddle2ONNX.git."); - } - - virtual ~Mapper() = default; - bool is_experimental_op_ = false; - const PaddleParser* parser_; - OnnxHelper* helper_; - int32_t block_idx_; - int32_t op_idx_; - std::string name_; // op transform name - - std::string OpType() const { - auto& op = parser_->GetOpDesc(block_idx_, op_idx_); - return op.type(); - } - - std::string Name() const { return name_; } - - bool HasInput(const std::string& name) const { - return parser_->OpHasInput(block_idx_, op_idx_, name); - } - bool HasOutput(const std::string& name) const { - return parser_->OpHasOutput(block_idx_, op_idx_, name); - } - std::vector GetInput(const std::string& name) const { - return parser_->GetOpInput(block_idx_, op_idx_, name); - } - std::vector GetOutput(const std::string& name) const { - return parser_->GetOpOutput(block_idx_, op_idx_, name); - } - // Judge whether Attribute(name)'s type is Var or Vars. - bool IsAttrVar(const std::string& name) const { - return parser_->OpIsAttrVar(block_idx_, op_idx_, name); - } - - // Get TensorInfo(s) from Attribute Var or Vars. - std::vector GetAttrVar(const std::string& name) const { - return parser_->GetOpAttrVar(block_idx_, op_idx_, name); - } - - bool HasAttr(const std::string& name) const { - auto& op = parser_->GetOpDesc(block_idx_, op_idx_); - return parser_->OpHasAttr(op, name); - } - void GetAttr(const std::string& name, int64_t* val) { - auto& op = parser_->GetOpDesc(block_idx_, op_idx_); - parser_->GetOpAttr(op, name, val); - } - void GetAttr(const std::string& name, float* val) { - auto& op = parser_->GetOpDesc(block_idx_, op_idx_); - parser_->GetOpAttr(op, name, val); - } - void GetAttr(const std::string& name, bool* val) { - auto& op = parser_->GetOpDesc(block_idx_, op_idx_); - parser_->GetOpAttr(op, name, val); - } - void GetAttr(const std::string& name, std::string* val) { - auto& op = parser_->GetOpDesc(block_idx_, op_idx_); - parser_->GetOpAttr(op, name, val); - } - void GetAttr(const std::string& name, std::vector* val) { - auto& op = parser_->GetOpDesc(block_idx_, op_idx_); - parser_->GetOpAttr(op, name, val); - } - void GetAttr(const std::string& name, std::vector* val) { - auto& op = parser_->GetOpDesc(block_idx_, op_idx_); - parser_->GetOpAttr(op, name, val); - } - void GetAttr(const std::string& name, std::vector* val) { - auto& op = parser_->GetOpDesc(block_idx_, op_idx_); - parser_->GetOpAttr(op, name, val); - } - - bool IsConstantInput(const std::string& input_key) const { - auto input_info = GetInput(input_key); - return parser_->IsConstantTensor(block_idx_, input_info[0].name); - } - - bool IsConstant(const TensorInfo& info) const { - return parser_->IsConstantTensor(block_idx_, info.name); - } - - template - bool TryGetInputValue(const std::string& input_key, std::vector* data) { - auto input_info = GetInput(input_key); - return parser_->TryGetTensorValue(block_idx_, input_info[0].name, data); - } - - template - bool TryGetValue(const TensorInfo& info, std::vector* data) { - return parser_->TryGetTensorValue(block_idx_, info.name, data); - } -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/affine_channel.cc b/paddle2onnx/mapper/nn/affine_channel.cc deleted file mode 100644 index 87b5efd802e..00000000000 --- a/paddle2onnx/mapper/nn/affine_channel.cc +++ /dev/null @@ -1,44 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/nn/affine_channel.h" - -namespace paddle2onnx { -REGISTER_MAPPER(affine_channel, AffineChannelMapper) - -int32_t AffineChannelMapper::GetMinOpset(bool verbose) { - if (data_layout_ == "NHWC") { - Error() << "Data format NHWC is not supported." << std::endl; - return false; - } - return 7; -} - -void AffineChannelMapper::Opset7() { - auto x_info = GetInput("X"); - auto scale_info = GetInput("Scale"); - auto bias_info = GetInput("Bias"); - auto out_info = GetOutput("Out"); - - auto scale = scale_info[0].name; - auto bias = bias_info[0].name; - if (scale_info[0].shape.size() <= 1) { - scale = helper_->Reshape(scale, {1, -1, 1, 1}); - bias = helper_->Reshape(bias, {1, -1, 1, 1}); - } - auto out = helper_->MakeNode("Mul", {x_info[0].name, scale})->output(0); - helper_->MakeNode("Add", {out, bias}, {out_info[0].name}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/affine_channel.h b/paddle2onnx/mapper/nn/affine_channel.h deleted file mode 100644 index fcb14920ccd..00000000000 --- a/paddle2onnx/mapper/nn/affine_channel.h +++ /dev/null @@ -1,38 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class AffineChannelMapper : public Mapper { - public: - AffineChannelMapper(const PaddleParser& p, OnnxHelper* helper, - int64_t block_id, int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("data_layout", &data_layout_); - } - - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - std::string data_layout_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/batch_norm.cc b/paddle2onnx/mapper/nn/batch_norm.cc deleted file mode 100644 index dd81622e087..00000000000 --- a/paddle2onnx/mapper/nn/batch_norm.cc +++ /dev/null @@ -1,45 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/nn/batch_norm.h" - -#include -#include - -namespace paddle2onnx { -REGISTER_MAPPER(batch_norm, BatchNormMapper) - -void BatchNormMapper::Opset7() { - auto input_info = GetInput("X"); - auto scale_info = GetInput("Scale"); - auto bias_info = GetInput("Bias"); - auto mean_info = GetInput("Mean"); - auto variance_info = GetInput("Variance"); - auto output_info = GetOutput("Y"); - - auto node = helper_->MakeNode( - "BatchNormalization", - {input_info[0].name, scale_info[0].name, bias_info[0].name, - mean_info[0].name, variance_info[0].name}, - {output_info[0].name}); - if (helper_->GetOpsetVersion() < 9) { - int64_t spatial = 1; - AddAttribute(node, "spatial", spatial); - } - - AddAttribute(node, "epsilon", epsilon_); - AddAttribute(node, "momentum", momentum_); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/batch_norm.h b/paddle2onnx/mapper/nn/batch_norm.h deleted file mode 100644 index abc649e0224..00000000000 --- a/paddle2onnx/mapper/nn/batch_norm.h +++ /dev/null @@ -1,39 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class BatchNormMapper : public Mapper { - public: - BatchNormMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("epsilon", &epsilon_); - GetAttr("momentum", &momentum_); - } - - void Opset7(); - - private: - float epsilon_; - float momentum_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/conv2d.cc b/paddle2onnx/mapper/nn/conv2d.cc deleted file mode 100644 index 7c763a0662c..00000000000 --- a/paddle2onnx/mapper/nn/conv2d.cc +++ /dev/null @@ -1,81 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/nn/conv2d.h" - -#include -#include - -namespace paddle2onnx { -REGISTER_MAPPER(conv2d, Conv2dMapper) -REGISTER_MAPPER(depthwise_conv2d, Conv2dMapper) - -int32_t Conv2dMapper::GetMinOpset(bool verbose) { - // NHWC is not supported - if (data_format_ == "NHWC") { - Error() << "Cannot support input with NHWC format." << std::endl; - return -1; - } - if (padding_algorithm_ == "EXPLICIT") { - if (paddings_.size() != 2 && paddings_.size() != 4) { - Error() << "While padding_algorithm is EXPLICIT, size of paddings should " - "be 2 or 4." - << std::endl; - return -1; - } - } - if (dilations_[0] != 1 || dilations_[1] != 1) { - if (padding_algorithm_ == "SAME") { - Error() << "While dilations != 1, cannot support padding = 'SAME'." - << std::endl; - return -1; - } - } - return 7; -} - -void Conv2dMapper::Opset7() { - auto kernel_info = GetInput("Filter"); - auto input_info = GetInput("Input"); - auto output_info = GetOutput("Output"); - - auto node = helper_->MakeNode( - "Conv", {input_info[0].name, kernel_info[0].name}, {output_info[0].name}); - AddAttribute(node, "dilations", dilations_); - std::vector kernel_shape = {kernel_info[0].shape[2], - kernel_info[0].shape[3]}; - AddAttribute(node, "kernel_shape", kernel_shape); - AddAttribute(node, "strides", strides_); - AddAttribute(node, "group", groups_); - if (padding_algorithm_ == "SAME") { - std::string auto_pad = "SAME_UPPER"; - AddAttribute(node, "auto_pad", auto_pad); - } else if (padding_algorithm_ == "VALID") { - std::string auto_pad = "VALID"; - AddAttribute(node, "auto_pad", auto_pad); - } else { - std::vector paddings; - if (paddings_.size() == 2) { - paddings.insert(paddings.begin(), paddings_.begin(), paddings_.end()); - paddings.insert(paddings.begin(), paddings_.begin(), paddings_.end()); - } else { - paddings.assign(paddings_.begin(), paddings_.end()); - paddings[1] = paddings_[2]; - paddings[2] = paddings_[1]; - } - AddAttribute(node, "pads", paddings); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/conv2d.h b/paddle2onnx/mapper/nn/conv2d.h deleted file mode 100644 index 06efdf4ef08..00000000000 --- a/paddle2onnx/mapper/nn/conv2d.h +++ /dev/null @@ -1,48 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class Conv2dMapper : public Mapper { - public: - Conv2dMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("groups", &groups_); - GetAttr("dilations", &dilations_); - GetAttr("strides", &strides_); - GetAttr("paddings", &paddings_); - GetAttr("padding_algorithm", &padding_algorithm_); - GetAttr("data_format", &data_format_); - } - - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - std::vector dilations_; - std::vector strides_; - std::vector paddings_; - std::string padding_algorithm_; - std::string data_format_; - int64_t groups_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/conv2d_transpose.cc b/paddle2onnx/mapper/nn/conv2d_transpose.cc deleted file mode 100755 index b397c6f978b..00000000000 --- a/paddle2onnx/mapper/nn/conv2d_transpose.cc +++ /dev/null @@ -1,66 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/nn/conv2d_transpose.h" - -#include -#include - -namespace paddle2onnx { -REGISTER_MAPPER(conv2d_transpose, Conv2dTransposeMapper) -REGISTER_MAPPER(depthwise_conv2d_transpose, Conv2dTransposeMapper) - -int32_t Conv2dTransposeMapper::GetMinOpset(bool verbose) { - // NHWC is not supported - if (data_format_ == "NHWC") { - Error() << "[ERROR] Cannot support NHWC format for operator " - "conv2d_transpose/depthwise_conv2d_transpose." - << std::endl; - return -1; - } - return 7; -} - -void Conv2dTransposeMapper::Opset7() { - auto kernel_info = GetInput("Filter"); - auto input_info = GetInput("Input"); - auto output_info = GetOutput("Output"); - auto input = helper_->AutoCast(input_info[0].name, input_info[0].dtype, - P2ODataType::FP32); - auto kernel = helper_->AutoCast(kernel_info[0].name, kernel_info[0].dtype, - P2ODataType::FP32); - auto node = helper_->MakeNode("ConvTranspose", {input, kernel}); - AddAttribute(node, "dilations", dilations_); - std::vector kernel_shape = {kernel_info[0].shape[2], - kernel_info[0].shape[3]}; - AddAttribute(node, "kernel_shape", kernel_shape); - AddAttribute(node, "strides", strides_); - AddAttribute(node, "group", groups_); - if (padding_algorithm_ == "SAME") { - std::string auto_pad = "SAME_UPPER"; - AddAttribute(node, "auto_pad", auto_pad); - } else if (padding_algorithm_ == "VALID") { - std::string auto_pad = "VALID"; - AddAttribute(node, "auto_pad", auto_pad); - } else { - AddAttribute(node, "pads", paddings_); - } - if (output_padding_.size() > 0) { - AddAttribute(node, "output_padding", output_padding_); - } - helper_->AutoCast(node->output(0), output_info[0].name, P2ODataType::FP32, - output_info[0].dtype); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/conv2d_transpose.h b/paddle2onnx/mapper/nn/conv2d_transpose.h deleted file mode 100755 index 3a8b472fd28..00000000000 --- a/paddle2onnx/mapper/nn/conv2d_transpose.h +++ /dev/null @@ -1,58 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class Conv2dTransposeMapper : public Mapper { - public: - Conv2dTransposeMapper(const PaddleParser& p, OnnxHelper* helper, - int64_t block_id, int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("groups", &groups_); - GetAttr("dilations", &dilations_); - GetAttr("strides", &strides_); - GetAttr("paddings", &paddings_); - GetAttr("padding_algorithm", &padding_algorithm_); - GetAttr("output_padding", &output_padding_); - GetAttr("data_format", &data_format_); - if (paddings_.size() == 2) { - paddings_.push_back(paddings_[0]); - paddings_.push_back(paddings_[1]); - } else if (paddings_.size() == 4) { - int32_t tmp = paddings_[1]; - paddings_[1] = paddings_[2]; - paddings_[2] = tmp; - } - } - - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - std::vector dilations_; - std::vector strides_; - std::vector paddings_; - std::vector output_padding_; - std::string padding_algorithm_; - std::string data_format_; - int64_t groups_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/conv3d.cc b/paddle2onnx/mapper/nn/conv3d.cc deleted file mode 100644 index 44e457f9b1b..00000000000 --- a/paddle2onnx/mapper/nn/conv3d.cc +++ /dev/null @@ -1,82 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/nn/conv3d.h" - -#include -#include - -namespace paddle2onnx { -REGISTER_MAPPER(conv3d, Conv3dMapper) - -int32_t Conv3dMapper::GetMinOpset(bool verbose) { - // NDHWC is not supported - if (data_format_ == "NDHWC") { - Error() << "Cannot support input with NDHWC format." << std::endl; - return -1; - } - if (padding_algorithm_ == "EXPLICIT") { - if (paddings_.size() != 3 && paddings_.size() != 6) { - Error() << "While padding_algorithm is EXPLICIT, size of paddings should " - "be 3 or 6." - << std::endl; - return -1; - } - } - if (dilations_[0] != 1 || dilations_[1] != 1 || dilations_[2] != 1) { - if (padding_algorithm_ == "SAME") { - Error() << "While dilations != 1, cannot support padding = 'SAME'." - << std::endl; - return -1; - } - } - return 7; -} - -void Conv3dMapper::Opset7() { - auto kernel_info = GetInput("Filter"); - auto input_info = GetInput("Input"); - auto output_info = GetOutput("Output"); - - auto node = helper_->MakeNode( - "Conv", {input_info[0].name, kernel_info[0].name}, {output_info[0].name}); - AddAttribute(node, "dilations", dilations_); - std::vector kernel_shape = {kernel_info[0].shape[2], - kernel_info[0].shape[3], - kernel_info[0].shape[4]}; - AddAttribute(node, "kernel_shape", kernel_shape); - AddAttribute(node, "strides", strides_); - AddAttribute(node, "group", groups_); - if (padding_algorithm_ == "SAME") { - std::string auto_pad = "SAME_UPPER"; - AddAttribute(node, "auto_pad", auto_pad); - } else if (padding_algorithm_ == "VALID") { - std::string auto_pad = "VALID"; - AddAttribute(node, "auto_pad", auto_pad); - } else { - std::vector paddings; - if (paddings_.size() == 3) { - paddings.insert(paddings.begin(), paddings_.begin(), paddings_.end()); - paddings.insert(paddings.begin(), paddings_.begin(), paddings_.end()); - } else { - std::vector index = {0, 2, 4, 1, 3, 5}; - for (auto &i : index) { - paddings.push_back(paddings_[i]); - } - } - AddAttribute(node, "pads", paddings); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/conv3d.h b/paddle2onnx/mapper/nn/conv3d.h deleted file mode 100755 index 310edd5cc03..00000000000 --- a/paddle2onnx/mapper/nn/conv3d.h +++ /dev/null @@ -1,48 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class Conv3dMapper : public Mapper { - public: - Conv3dMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("groups", &groups_); - GetAttr("dilations", &dilations_); - GetAttr("strides", &strides_); - GetAttr("paddings", &paddings_); - GetAttr("padding_algorithm", &padding_algorithm_); - GetAttr("data_format", &data_format_); - } - - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - std::vector dilations_; - std::vector strides_; - std::vector paddings_; - std::string padding_algorithm_; - std::string data_format_; - int64_t groups_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/data_norm.cc b/paddle2onnx/mapper/nn/data_norm.cc deleted file mode 100644 index 2cea544f4c5..00000000000 --- a/paddle2onnx/mapper/nn/data_norm.cc +++ /dev/null @@ -1,43 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/nn/data_norm.h" - -namespace paddle2onnx { -REGISTER_MAPPER(data_norm, DataNormMapper) - -int32_t DataNormMapper::GetMinOpset(bool verbose) { - if (slot_dim_ > 0) { - Error() << "slot_dim > 0 is not supported." << std::endl; - return -1; - } - return 7; -} - -void DataNormMapper::Opset7() { - auto input_info = GetInput("X"); - auto batch_size_info = GetInput("BatchSize"); - auto batch_sum_info = GetInput("BatchSum"); - auto batch_square_sum_info = GetInput("BatchSquareSum"); - auto output_info = GetOutput("Y"); - - Assert(slot_dim_ <= 0, "slot_dim > 0 is not supported."); - auto mean_arr = helper_->MakeNode("Div", {batch_sum_info[0].name, batch_size_info[0].name})->output(0); - auto scale_arr = helper_->MakeNode("Div", {batch_size_info[0].name, batch_square_sum_info[0].name})->output(0); - scale_arr = helper_->MakeNode("Sqrt", {scale_arr})->output(0); - auto out = helper_->MakeNode("Sub", {input_info[0].name, mean_arr})->output(0); - helper_->MakeNode("Mul" ,{out, scale_arr}, {output_info[0].name}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/data_norm.h b/paddle2onnx/mapper/nn/data_norm.h deleted file mode 100644 index 576a53d9dc2..00000000000 --- a/paddle2onnx/mapper/nn/data_norm.h +++ /dev/null @@ -1,44 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class DataNormMapper : public Mapper { - public: - DataNormMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("data_layout", &data_layout_); - GetAttr("epsilon", &epsilon_); - if (HasAttr("slot_dim")) { - GetAttr("slot_dim", &slot_dim_); - } - } - - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - std::string data_layout_; - float epsilon_; - int64_t slot_dim_ = -1; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/dropout.cc b/paddle2onnx/mapper/nn/dropout.cc deleted file mode 100644 index 615b71984dd..00000000000 --- a/paddle2onnx/mapper/nn/dropout.cc +++ /dev/null @@ -1,65 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. -#include "paddle2onnx/mapper/nn/dropout.h" - -#include - -namespace paddle2onnx { - -REGISTER_MAPPER(dropout, DropoutMapper) - -int32_t DropoutMapper::GetMinOpset(bool verbose) { - if (dropout_implementation_ != "downgrade_in_infer" && - dropout_implementation_ != "upscale_in_train") { - Error() << "Drop out type: " << dropout_implementation_ - << " is not supported yet." << std::endl; - return -1; - } - if (dropout_implementation_ == "downgrade_in_infer") { - if (IsAttrVar("dropout_prob") && - !IsConstant(GetAttrVar("dropout_prob")[0])) { - Error() << "While Attribute(dropout_prob)'s type is Tensor, it's not " - "supported " - "unless it's a constant tensor when dropout_implementation is " - "downgrade_in_infer." - << std::endl; - return -1; - } - } - return 7; -} - -void DropoutMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - if (dropout_implementation_ == "upscale_in_train") { - helper_->MakeNode("Identity", {input_info[0].name}, {output_info[0].name}); - } else { - if (IsAttrVar("dropout_prob")) { - auto prob_info = GetAttrVar("dropout_prob"); - std::vector temp; - TryGetValue(prob_info[0], &temp); - dropout_prob_ = temp[0]; - } else { - GetAttr("dropout_prob", &dropout_prob_); - } - std::string scale_node = helper_->Constant( - {}, GetOnnxDtype(input_info[0].dtype), 1 - dropout_prob_); - helper_->MakeNode("Mul", {input_info[0].name, scale_node}, - {output_info[0].name}); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/dropout.h b/paddle2onnx/mapper/nn/dropout.h deleted file mode 100755 index 048d9b6603c..00000000000 --- a/paddle2onnx/mapper/nn/dropout.h +++ /dev/null @@ -1,37 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. -#pragma once -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class DropoutMapper : public Mapper { - public: - DropoutMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("dropout_implementation", &dropout_implementation_); - } - - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - float dropout_prob_ = 0.0; - std::string dropout_implementation_ = "upscale_in_train"; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/group_norm.cc b/paddle2onnx/mapper/nn/group_norm.cc deleted file mode 100644 index 717401d270b..00000000000 --- a/paddle2onnx/mapper/nn/group_norm.cc +++ /dev/null @@ -1,75 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/nn/group_norm.h" - -#include -#include -#include - -namespace paddle2onnx { -REGISTER_MAPPER(group_norm, GroupNormMapper) - -int32_t GroupNormMapper::GetMinOpset(bool verbose) { - auto input_info = GetInput("X"); - if (input_info[0].Rank() != 4) { - Error() << "Only support 4D-Tensor as input for GroupNorm" << std::endl; - return -1; - } - return 7; -} - -void GroupNormMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Y"); - - std::vector shape_val = {0, groups_, -1}; - std::string shape = - helper_->Constant(GetOnnxDtype(P2ODataType::INT64), shape_val); - - auto reshape_input = - helper_->MakeNode("Reshape", {input_info[0].name, shape}); - - std::string scale_ = helper_->Constant(GetOnnxDtype(input_info[0].dtype), - std::vector(groups_, 1.0)); - - std::string bias_ = helper_->Constant(GetOnnxDtype(input_info[0].dtype), - std::vector(groups_, 0.0)); - - auto reshaped_output = helper_->MakeNode( - "InstanceNormalization", {reshape_input->output(0), scale_, bias_}); - AddAttribute(reshaped_output, "epsilon", epsilon_); - - auto origin_shape = helper_->MakeNode("Shape", {input_info[0].name}); - - if (HasInput("Scale") && HasInput("Bias")) { - auto scale_info = GetInput("Scale"); - auto bias_info = GetInput("Bias"); - auto output = helper_->MakeNode( - "Reshape", {reshaped_output->output(0), origin_shape->output(0)}); - std::string unsqueezed_scale = - helper_->Unsqueeze(scale_info[0].name, {1, 2}); - std::string unsqueezed_bias = helper_->Unsqueeze(bias_info[0].name, {1, 2}); - auto scale_output = - helper_->MakeNode("Mul", {output->output(0), unsqueezed_scale}); - helper_->MakeNode("Add", {scale_output->output(0), unsqueezed_bias}, - {output_info[0].name}); - } else { - helper_->MakeNode("Reshape", - {reshaped_output->output(0), origin_shape->output(0)}, - {output_info[0].name}); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/group_norm.h b/paddle2onnx/mapper/nn/group_norm.h deleted file mode 100755 index a2bd1bd8d60..00000000000 --- a/paddle2onnx/mapper/nn/group_norm.h +++ /dev/null @@ -1,40 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class GroupNormMapper : public Mapper { - public: - GroupNormMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("groups", &groups_); - GetAttr("epsilon", &epsilon_); - } - - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - int64_t groups_; - float epsilon_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/instance_norm.cc b/paddle2onnx/mapper/nn/instance_norm.cc deleted file mode 100644 index edddac10f4e..00000000000 --- a/paddle2onnx/mapper/nn/instance_norm.cc +++ /dev/null @@ -1,59 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/nn/instance_norm.h" - -#include -#include - -namespace paddle2onnx { -REGISTER_MAPPER(instance_norm, InstanceNormMapper) - -int32_t InstanceNormMapper::GetMinOpset(bool verbose) { - auto input_info = GetInput("X"); - int num_groups = input_info[0].shape[1]; - if (num_groups < 0) { - Error() << "The dimension in axis=1 of input tensor must be known, but now it's unknown." << std::endl; - return -1; - } - return 7; -} - -void InstanceNormMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Y"); - int num_groups = input_info[0].shape[1]; - - std::string scale = ""; - if (HasInput("Scale")) { - scale = GetInput("Scale")[0].name; - } else { - scale = helper_->Constant(GetOnnxDtype(input_info[0].dtype), std::vector(num_groups, 1.0)); - } - - std::string bias = ""; - if (HasInput("Bias")) { - bias = GetInput("Bias")[0].name; - } else { - bias = helper_->Constant(GetOnnxDtype(input_info[0].dtype), std::vector(num_groups, 0.0)); - } - - auto node = helper_->MakeNode( - "InstanceNormalization", - {input_info[0].name, scale, bias}, - {output_info[0].name}); - AddAttribute(node, "epsilon", epsilon_); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/instance_norm.h b/paddle2onnx/mapper/nn/instance_norm.h deleted file mode 100644 index c569b1f92c1..00000000000 --- a/paddle2onnx/mapper/nn/instance_norm.h +++ /dev/null @@ -1,38 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class InstanceNormMapper : public Mapper { - public: - InstanceNormMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("epsilon", &epsilon_); - } - - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - float epsilon_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/interpolate.cc b/paddle2onnx/mapper/nn/interpolate.cc deleted file mode 100755 index 49523a8a69a..00000000000 --- a/paddle2onnx/mapper/nn/interpolate.cc +++ /dev/null @@ -1,136 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/nn/interpolate.h" - -namespace paddle2onnx { -REGISTER_MAPPER(bilinear_interp, InterpolateMapper) -REGISTER_MAPPER(bilinear_interp_v2, InterpolateMapper) -REGISTER_MAPPER(nearest_interp_v2, InterpolateMapper) -REGISTER_MAPPER(bicubic_interp_v2, InterpolateMapper) -REGISTER_MAPPER(linear_interp_v2, InterpolateMapper) -REGISTER_MAPPER(trilinear_interp_v2, InterpolateMapper) - -int32_t InterpolateMapper::GetMinOpset(bool verbose) { - if (data_layout_ == "NHWC") { - Error() << "Data format of NHWC is not supported." << std::endl; - return -1; - } - auto x_info = GetInput("X"); - if (x_info[0].Rank() > 5 && x_info[0].Rank() < 3) { - Error() << "Only support 3D/4D/5D tensor, but now its dimension is " - << x_info[0].Rank() << std::endl; - return -1; - } - Logger(verbose, 11) << RequireOpset(11) << std::endl; - return 11; -} - -std::string InterpolateMapper::ComputeOutSize() { - bool has_out_size = HasInput("OutSize"); - bool has_size_tensor = HasInput("SizeTensor"); - if (has_out_size) { - auto out_size_info = GetInput("OutSize"); - return helper_->AutoCast(out_size_info[0].name, out_size_info[0].dtype, - P2ODataType::INT64); - } else { - auto size_tensor_info = GetInput("SizeTensor"); - return helper_->ConcatIndices(size_tensor_info); - } -} - -std::string InterpolateMapper::ComputeScale() { - auto scale_info = GetInput("Scale"); - auto scale = helper_->AutoCast(scale_info[0].name, scale_info[0].dtype, - P2ODataType::FP32); - auto padding = helper_->Constant(ONNX_NAMESPACE::TensorProto::FLOAT, - std::vector(2, 1.0)); - scale = helper_->Concat({padding, scale}, 0); - return scale; -} - -void InterpolateMapper::Opset11() { - auto x_info = GetInput("X"); - auto out_info = GetOutput("Out"); - std::string coordinate_transformation_mode = "half_pixel"; - auto resize_type = resize_mapper_[method_]; - if (align_corners_) { - coordinate_transformation_mode = "align_corners"; - } else if (resize_type == "nearest") { - coordinate_transformation_mode = "asymmetric"; - } else if (align_mode_ == 1 && resize_type != "cubic") { - coordinate_transformation_mode = "asymmetric"; - } - std::string scale = ""; - std::string size = ""; - bool has_out_size = HasInput("OutSize"); - bool has_size_tensor = HasInput("SizeTensor"); - bool has_scale_tensor = HasInput("Scale"); - if (has_out_size || has_size_tensor) { - size = ComputeOutSize(); - } else if (has_scale_tensor) { - scale = ComputeScale(); - } else { - // get size or scale from attribute - if (out_d_ > 0 || out_w_ > 0 || out_h_ > 0) { - std::vector out_size; - if (x_info[0].Rank() == 5) { - out_size.push_back(out_d_); - out_size.push_back(out_h_); - } - if (x_info[0].Rank() == 4) { - out_size.push_back(out_h_); - } - out_size.push_back(out_w_); - size = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, out_size); - } else { - std::vector scale_; - GetAttr("scale", &scale_); - float padding = 1.0; - scale_.insert(scale_.begin(), padding); - scale_.insert(scale_.begin(), padding); - scale = helper_->Constant(ONNX_NAMESPACE::TensorProto::FLOAT, scale_); - } - } - std::string roi = helper_->Constant(ONNX_NAMESPACE::TensorProto::FLOAT, std::vector()); - if (scale == "") { - // has to generate a empty tensor for resize - scale = helper_->Constant(ONNX_NAMESPACE::TensorProto::FLOAT, - std::vector()); - } - if (size != "") { - auto ipt_shape = helper_->MakeNode("Shape", {x_info[0].name})->output(0); - auto nc = helper_->Slice(ipt_shape, {0}, {0}, {2}); - size = helper_->Concat({nc, size}, 0); - } - std::shared_ptr node; - if (size != "") { - node = helper_->MakeNode("Resize", {x_info[0].name, roi, scale, size}, - {out_info[0].name}); - } else { - node = helper_->MakeNode("Resize", {x_info[0].name, roi, scale}, - {out_info[0].name}); - } - Assert(resize_mapper_.find(OpType()) != resize_mapper_.end(), - "Cannot find " + OpType() + " in resize_mapper."); - AddAttribute(node, "mode", resize_mapper_[OpType()]); - AddAttribute(node, "coordinate_transformation_mode", - coordinate_transformation_mode); - if (resize_mapper_[OpType()] == "nearest" && - coordinate_transformation_mode == "asymmetric") { - AddAttribute(node, "nearest_mode", "floor"); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/interpolate.h b/paddle2onnx/mapper/nn/interpolate.h deleted file mode 100755 index d3b8a7a7a88..00000000000 --- a/paddle2onnx/mapper/nn/interpolate.h +++ /dev/null @@ -1,57 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class InterpolateMapper : public Mapper { - public: - InterpolateMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("data_layout", &data_layout_); - GetAttr("align_corners", &align_corners_); - GetAttr("align_mode", &align_mode_); - GetAttr("out_d", &out_d_); - GetAttr("out_h", &out_h_); - GetAttr("out_w", &out_w_); - method_ = OpType(); - - resize_mapper_["bilinear_interp"] = "linear"; - resize_mapper_["bilinear_interp_v2"] = "linear"; - resize_mapper_["nearest_interp_v2"] = "nearest"; - resize_mapper_["bicubic_interp_v2"] = "cubic"; - resize_mapper_["linear_interp_v2"] = "linear"; - resize_mapper_["trilinear_interp_v2"] = "linear"; - } - - int32_t GetMinOpset(bool verbose = false); - void Opset11(); - - private: - std::string ComputeOutSize(); - std::string ComputeScale(); - std::map resize_mapper_; - std::string method_; - std::string data_layout_; - int64_t align_mode_; - int64_t out_d_; - int64_t out_h_; - int64_t out_w_; - bool align_corners_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/layer_norm.cc b/paddle2onnx/mapper/nn/layer_norm.cc deleted file mode 100644 index 5fab8da22ca..00000000000 --- a/paddle2onnx/mapper/nn/layer_norm.cc +++ /dev/null @@ -1,146 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/nn/layer_norm.h" - -#include -#include -#include - -namespace paddle2onnx { -REGISTER_MAPPER(layer_norm, LayerNormMapper) - -void LayerNormMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Y"); - - std::string input_name = helper_->AutoCast( - input_info[0].name, input_info[0].dtype, P2ODataType::FP32); - - std::vector input_shape = input_info[0].shape; - std::vector axes; - for (auto i = begin_norm_axis_; i < input_shape.size(); i++) { - axes.push_back(i); - } - if (begin_norm_axis_ == input_shape.size() - 1) { - axes[0] = -1; - } - - float epsilon = epsilon_; - std::string epsilon_node = - helper_->Constant({}, GetOnnxDtype(P2ODataType::FP32), epsilon); - std::string two_node = - helper_->Constant({}, GetOnnxDtype(P2ODataType::FP32), float(2.0)); - - auto mean_node = helper_->MakeNode("ReduceMean", {input_name}); - AddAttribute(mean_node, "axes", axes); - - auto numerator_node = - helper_->MakeNode("Sub", {input_name, mean_node->output(0)}); - auto pow_num_node = - helper_->MakeNode("Pow", {numerator_node->output(0), two_node}); - - auto variance_node = - helper_->MakeNode("ReduceMean", {pow_num_node->output(0)}); - AddAttribute(variance_node, "axes", axes); - - auto add_eps_node = - helper_->MakeNode("Add", {variance_node->output(0), epsilon_node}); - - auto denominator_node = helper_->MakeNode("Sqrt", {add_eps_node->output(0)}); - - auto ipt_shape_node = helper_->MakeNode("Shape", {input_name}); - std::vector slice_axes = {0}; - std::vector start = { - static_cast(input_shape.size() - axes.size())}; - std::vector end = {static_cast(input_shape.size())}; - std::string weight_shape_node = - helper_->Slice(ipt_shape_node->output(0), slice_axes, start, end); - - bool has_input_Bias = HasInput("Bias"); - bool has_input_Scale = HasInput("Scale"); - - if (has_input_Bias && has_input_Scale) { - auto scale_info = GetInput("Scale"); - auto bias_info = GetInput("Bias"); - std::string scale_name = helper_->AutoCast( - scale_info[0].name, scale_info[0].dtype, P2ODataType::FP32); - std::string bias_name = helper_->AutoCast( - bias_info[0].name, bias_info[0].dtype, P2ODataType::FP32); - std::string scale_node = ""; - std::string bias_node = ""; - if (begin_norm_axis_ == input_shape.size() - 1) { - scale_node = helper_->Reshape(scale_name, {-1}); - bias_node = helper_->Reshape(bias_name, {-1}); - } else { - scale_node = helper_->MakeNode("Reshape", {scale_name, weight_shape_node}) - ->output(0); - bias_node = helper_->MakeNode("Reshape", {bias_name, weight_shape_node}) - ->output(0); - } - auto layer_norm_pre_node = helper_->MakeNode( - "Div", {numerator_node->output(0), denominator_node->output(0)}); - auto layer_norm_node = - helper_->MakeNode("Mul", {layer_norm_pre_node->output(0), scale_node}); - auto pre_cast_node = - helper_->MakeNode("Add", {layer_norm_node->output(0), bias_node}); - helper_->AutoCast(pre_cast_node->output(0), output_info[0].name, - P2ODataType::FP32, output_info[0].dtype); - return; - } - if (has_input_Bias) { - auto bias_info = GetInput("Bias"); - std::string bias_name = helper_->AutoCast( - bias_info[0].name, bias_info[0].dtype, P2ODataType::FP32); - std::string bias_node = ""; - if (begin_norm_axis_ == input_shape.size() - 1) { - bias_node = helper_->Reshape(bias_name, {-1}); - } else { - bias_node = helper_->MakeNode("Reshape", {bias_name, weight_shape_node}) - ->output(0); - } - auto layer_norm_node = helper_->MakeNode( - "Div", {numerator_node->output(0), denominator_node->output(0)}); - auto pre_cast_node = - helper_->MakeNode("Add", {layer_norm_node->output(0), bias_node}); - helper_->AutoCast(pre_cast_node->output(0), output_info[0].name, - P2ODataType::FP32, output_info[0].dtype); - return; - } - if (has_input_Scale) { - auto scale_info = GetInput("Scale"); - std::string scale_name = helper_->AutoCast( - scale_info[0].name, scale_info[0].dtype, P2ODataType::FP32); - std::string scale_node = ""; - if (begin_norm_axis_ == input_shape.size() - 1) { - scale_node = helper_->Reshape(scale_name, {-1}); - } else { - scale_node = helper_->MakeNode("Reshape", {scale_name, weight_shape_node}) - ->output(0); - } - auto layer_norm_node = helper_->MakeNode( - "Div", {numerator_node->output(0), denominator_node->output(0)}); - auto pre_cast_node = - helper_->MakeNode("Mul", {layer_norm_node->output(0), scale_node}); - helper_->AutoCast(pre_cast_node->output(0), output_info[0].name, - P2ODataType::FP32, output_info[0].dtype); - return; - } - auto pre_cast_node = helper_->MakeNode( - "Div", {numerator_node->output(0), denominator_node->output(0)}); - helper_->AutoCast(pre_cast_node->output(0), output_info[0].name, - P2ODataType::FP32, output_info[0].dtype); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/layer_norm.h b/paddle2onnx/mapper/nn/layer_norm.h deleted file mode 100644 index 6eebe062e0c..00000000000 --- a/paddle2onnx/mapper/nn/layer_norm.h +++ /dev/null @@ -1,39 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class LayerNormMapper : public Mapper { - public: - LayerNormMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("begin_norm_axis", &begin_norm_axis_); - GetAttr("epsilon", &epsilon_); - } - - void Opset7(); - - private: - int64_t begin_norm_axis_; - float epsilon_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/norm.cc b/paddle2onnx/mapper/nn/norm.cc deleted file mode 100755 index abe188f2749..00000000000 --- a/paddle2onnx/mapper/nn/norm.cc +++ /dev/null @@ -1,28 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/nn/norm.h" - -namespace paddle2onnx { -REGISTER_MAPPER(norm, NormMapper) - -void NormMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - auto node = helper_->MakeNode("LpNormalization", {input_info[0].name}, - {output_info[0].name}); - AddAttribute(node, "axis", axis_); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/norm.h b/paddle2onnx/mapper/nn/norm.h deleted file mode 100644 index f3eb01bfd5f..00000000000 --- a/paddle2onnx/mapper/nn/norm.h +++ /dev/null @@ -1,37 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class NormMapper : public Mapper { - public: - NormMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("axis", &axis_); - } - - void Opset7(); - - private: - int64_t axis_ = -1; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/pad.cc b/paddle2onnx/mapper/nn/pad.cc deleted file mode 100644 index 0dd9d94f469..00000000000 --- a/paddle2onnx/mapper/nn/pad.cc +++ /dev/null @@ -1,54 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/nn/pad.h" - -namespace paddle2onnx { -REGISTER_MAPPER(pad, PadMapper) - -std::vector PadMapper::ConvertPaddingParameter( - const std::vector& paddings) { - std::vector new_paddings(paddings.size(), 0); - Assert(paddings.size() % 2 == 0, "The size of padding should be even"); - int64_t half_paddings_len = paddings.size() / 2; - for (auto i = 0; i < half_paddings_len; ++i) { - new_paddings[i] = paddings[2 * i]; - new_paddings[i + half_paddings_len] = paddings[2 * i + 1]; - } - return new_paddings; -} - -void PadMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - auto node = - helper_->MakeNode("Pad", {input_info[0].name}, {output_info[0].name}); - AddAttribute(node, "mode", "constant"); - AddAttribute(node, "value", pad_value_); - AddAttribute(node, "pads", ConvertPaddingParameter(paddings_)); -} - -void PadMapper::Opset11() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - auto paddings = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, - ConvertPaddingParameter(paddings_)); - auto value = - helper_->Constant({}, GetOnnxDtype(input_info[0].dtype), pad_value_); - auto node = helper_->MakeNode("Pad", {input_info[0].name, paddings, value}, - {output_info[0].name}); - AddAttribute(node, "mode", "constant"); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/pad.h b/paddle2onnx/mapper/nn/pad.h deleted file mode 100644 index d4813d03b5f..00000000000 --- a/paddle2onnx/mapper/nn/pad.h +++ /dev/null @@ -1,41 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class PadMapper : public Mapper { - public: - PadMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("pad_value", &pad_value_); - GetAttr("paddings", &paddings_); - } - void Opset7(); - void Opset11(); - - private: - std::vector ConvertPaddingParameter( - const std::vector& paddings); - std::vector paddings_; - float pad_value_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/pad3d.cc b/paddle2onnx/mapper/nn/pad3d.cc deleted file mode 100644 index 3040247364e..00000000000 --- a/paddle2onnx/mapper/nn/pad3d.cc +++ /dev/null @@ -1,114 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/nn/pad3d.h" - -namespace paddle2onnx { -REGISTER_MAPPER(pad3d, Pad3DMapper) - -int32_t Pad3DMapper::GetMinOpset(bool verbose) { - if (data_format_ == "NDHWC") { - Error() << "NDHWC format is not supported." << std::endl; - return -1; - } - if (mode_ == "circular") { - Error() << "Padding mode `circular` is not supported." << std::endl; - return -1; - } - if (HasInput("Paddings")) { - if (!IsConstantInput("Paddings")) { - Logger(verbose, 11) << "While Paddings is input and it's not a constant tensor, " << RequireOpset(11) << std::endl; - return 11; - } - std::vector paddings; - if (!TryGetInputValue("Paddings", &paddings)) { - Logger(verbose, 11) << "Cannot get constant value from input of Paddings, " << RequireOpset(11) << std::endl; - return 11; - } else { - if (paddings.size() != 6) { - Error() << "Size of paddings should be equal to 6, but now it's " << paddings.size() << std::endl; - return -1; - } - } - } else { - if (paddings_.size() != 6) { - Error() << "Size of paddings should be equal to 6, but now it's " << paddings_.size() << std::endl; - return -1; - } - } - return 7; -} - -std::vector Pad3DMapper::ConvertPaddingParameter(const std::vector& paddings) { - std::vector new_paddings(10, 0); - new_paddings[2] = paddings[4]; - new_paddings[3] = paddings[2]; - new_paddings[4] = paddings[0]; - new_paddings[7] = paddings[5]; - new_paddings[8] = paddings[3]; - new_paddings[9] = paddings[1]; - return new_paddings; -} - -void Pad3DMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - auto mode = mode_; - if (mode == "replicate") { - mode = "edge"; - } - std::vector paddings; - if (HasInput("Paddings")) { - Assert(TryGetInputValue("Paddings", &paddings), "Cannot get constant value from input of Paddings, " + RequireOpset(11)); - } else { - paddings.assign(paddings_.begin(), paddings_.end()); - } - std::vector new_paddings = ConvertPaddingParameter(paddings); - auto node = helper_->MakeNode("Pad", {input_info[0].name}, {output_info[0].name}); - AddAttribute(node, "mode", mode); - AddAttribute(node, "value", value_); - AddAttribute(node, "pads", new_paddings); -} - -void Pad3DMapper::Opset11() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - auto mode = mode_; - if (mode == "replicate") { - mode = "edge"; - } - - std::string paddings = ""; - if (HasInput("Paddings")) { - std::vector paddings_value; - if (TryGetInputValue("Paddings", &paddings_value)) { - std::vector new_paddings = ConvertPaddingParameter(paddings_value); - paddings = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, new_paddings); - } else { - auto pad_info = GetInput("Paddings"); - auto cast_pad = helper_->AutoCast(pad_info[0].name, pad_info[0].dtype, P2ODataType::INT64); - auto split_pads = helper_->Split(cast_pad, std::vector(6, 1), 0); - auto zero = helper_->Constant({1}, ONNX_NAMESPACE::TensorProto::INT64, int64_t(0)); - paddings = helper_->Concat({zero, zero, split_pads[4], split_pads[2], split_pads[0], zero, zero, split_pads[5], split_pads[3], split_pads[1]}, 0); - } - } else { - std::vector new_paddings = ConvertPaddingParameter(paddings_); - paddings = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, new_paddings); - } - auto value = helper_->Constant({}, GetOnnxDtype(input_info[0].dtype), value_); - auto node = helper_->MakeNode("Pad", {input_info[0].name, paddings, value}, {output_info[0].name}); - AddAttribute(node, "mode", mode); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/pad3d.h b/paddle2onnx/mapper/nn/pad3d.h deleted file mode 100644 index 3e9f50599d8..00000000000 --- a/paddle2onnx/mapper/nn/pad3d.h +++ /dev/null @@ -1,45 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class Pad3DMapper : public Mapper { - public: - Pad3DMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("data_format", &data_format_); - GetAttr("mode", &mode_); - GetAttr("value", &value_); - GetAttr("paddings", &paddings_); - } - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - void Opset11(); - - private: - std::vector ConvertPaddingParameter(const std::vector& paddings); - std::string data_format_; - std::string mode_; - std::vector paddings_; - float value_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/pool2d.cc b/paddle2onnx/mapper/nn/pool2d.cc deleted file mode 100755 index 0996bca8a28..00000000000 --- a/paddle2onnx/mapper/nn/pool2d.cc +++ /dev/null @@ -1,332 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/nn/pool2d.h" - -#include -#include -#include -#include - -namespace paddle2onnx { -REGISTER_MAPPER(pool2d, Pool2dMapper) -REGISTER_MAPPER(max_pool2d_with_index, Pool2dMapper) - -bool Pool2dMapper::IsSameSpan(const int64_t& in_size, const int64_t& out_size) { - std::vector spans; - spans.reserve(out_size); - for (auto i = 0; i < out_size; ++i) { - int64_t start = std::floor(i * (in_size / out_size)); - int64_t end = std::ceil((i + 1) * (in_size / out_size)); - spans.push_back(end - start); - } - std::sort(spans.begin(), spans.end()); - return spans[0] == spans[spans.size() - 1]; -} - -bool Pool2dMapper::IsExportAsCustomOp() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - GetAttr("ksize", &k_size_); - if (global_pooling_ || (k_size_[0] == 1 && k_size_[1] == 1)) { - return false; - } - if (export_as_custom_op && adaptive_) { - bool is_1x1_kernel = true; - for (auto i : k_size_) { - if (i != 1) { - is_1x1_kernel = false; - } - } - if (is_1x1_kernel) { - return false; - } - for (auto one_input : input_info) { - for (auto i = 2; i < one_input.shape.size(); ++i) { - if (one_input.shape[i] == -1) { - return true; - } - } - } - int64_t input_h = input_info[0].shape[2]; - int64_t input_w = input_info[0].shape[3]; - int64_t output_h = output_info[0].shape[2]; - int64_t output_w = output_info[0].shape[3]; - if (output_h == -1 || output_w == -1 || !IsSameSpan(input_h, output_h) || - !IsSameSpan(input_w, output_w)) { - return true; - } - } - return false; -} - -void Pool2dMapper::ExportAsCustomOp() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - auto node = helper_->MakeNode(custom_op_name, {input_info[0].name}, - {output_info[0].name}); - node->set_domain("Paddle"); - AddAttribute(node, "pooling_type", pooling_type_); - for (auto i = 1; i < output_info[0].shape.size(); i++) { - if (output_info[0].shape[i] == -1) { - if (input_info[0].shape[i] == -1) { - Assert(false, - "Can not convert to AdaptivePool custom OP, because the shapes " - "of the input and output are unknown."); - } else { - output_info[0].shape[i] = input_info[0].shape[i]; - } - } - } - AddAttribute(node, "output_size", output_info[0].shape); - Warn() << "Pool2d is exported as custom operator: " << custom_op_name - << std::endl; - helper_->MakeValueInfo(input_info[0].name, input_info[0].dtype, - input_info[0].shape); - helper_->MakeValueInfo(output_info[0].name, output_info[0].dtype, - output_info[0].shape); -} - -void Pool2dMapper::AdaptivePool(const std::vector& input_info, - const std::vector& output_info) { - int64_t input_h = input_info[0].shape[2]; - int64_t input_w = input_info[0].shape[3]; - int64_t output_h = output_info[0].shape[2]; - int64_t output_w = output_info[0].shape[3]; - int64_t stride_h = std::floor(input_h / output_h); - int64_t stride_w = std::floor(input_w / output_w); - int64_t kernel_h = input_h - (output_h - 1) * stride_h; - int64_t kernel_w = input_w - (output_w - 1) * stride_w; - std::string onnx_pool_type; - if (OpType() == "max_pool2d_with_index") { - onnx_pool_type = "MaxPool"; - } else { - auto iter = op_mapper_.find(pooling_type_); - onnx_pool_type = iter->second[0]; - } - - std::shared_ptr* node_ptr; - auto input = helper_->AutoCast(input_info[0].name, input_info[0].dtype, - P2ODataType::FP32); - auto node = helper_->MakeNode(onnx_pool_type, {input}); - helper_->AutoCast(node->output(0), output_info[0].name, P2ODataType::FP32, - output_info[0].dtype); - std::vector kernel_size = {kernel_h, kernel_w}; - AddAttribute(node, "kernel_shape", kernel_size); - std::vector strides = {stride_h, stride_w}; - AddAttribute(node, "strides", strides); - - if (helper_->GetOpsetVersion() > 10) { - AddAttribute(node, "ceil_mode", static_cast(ceil_mode_)); - } - - std::string auto_pad = "NOTSET"; - if (padding_algorithm_ == "SAME") { - auto_pad = "SAME_UPPER"; - } else if (padding_algorithm_ == "VALID") { - auto_pad = "VALID"; - } - AddAttribute(node, "auto_pad", auto_pad); - if (pooling_type_ == "avg") { - AddAttribute(node, "count_include_pad", static_cast(exclusive_)); - } -} - -void Pool2dMapper::NoAdaptivePool(const std::vector& input_info, - const std::vector& output_info) { - std::vector input_shape = input_info[0].shape; - if (pads_.size() == 2) { - pads_.push_back(pads_[0]); - pads_.push_back(pads_[1]); - } else if (pads_.size() == 4) { - std::vector index = {0, 2, 1, 3}; - std::vector copy = pads_; - for (auto i = 0; i < index.size(); ++i) { - pads_[i] = copy[index[i]]; - } - } - if (input_shape[2] > 0 && input_shape[2] + pads_[0] + pads_[2] < k_size_[0]) { - k_size_[0] = input_shape[2] + pads_[0] + pads_[2]; - } - if (input_shape[3] > 0 && input_shape[3] + pads_[1] + pads_[3] < k_size_[1]) { - k_size_[1] = input_shape[3] + pads_[1] + pads_[3]; - } - - int64_t max_ksize = *std::max_element(std::begin(k_size_), std::end(k_size_)); - int64_t max_pads = *std::max_element(std::begin(pads_), std::end(pads_)); - auto input_x = helper_->AutoCast(input_info[0].name, input_info[0].dtype, - P2ODataType::FP32); - if (max_ksize <= max_pads) { - std::vector onnx_paddings = {0, 0, pads_[0], pads_[1], - 0, 0, pads_[2], pads_[3]}; - std::vector inputs_names = {input_x}; - if (helper_->GetOpsetVersion() >= 11) { - std::string paddings_node = - helper_->Constant(GetOnnxDtype(P2ODataType::INT64), onnx_paddings); - inputs_names.push_back(paddings_node); - std::vector val = {0.0}; - std::string val_node = - helper_->Constant(GetOnnxDtype(P2ODataType::FP32), val); - inputs_names.push_back(val_node); - } - auto node = helper_->MakeNode("Pad", inputs_names); - std::string mode = "constant"; - AddAttribute(node, "mode", mode); - if (helper_->GetOpsetVersion() < 11) { - AddAttribute(node, "pads", onnx_paddings); - float val = 0.0; - AddAttribute(node, "value", val); - } - input_x = node->output(0); - pads_.clear(); - pads_.resize(4, 0); - } - std::string onnx_pool_type; - if (OpType() == "max_pool2d_with_index") { - onnx_pool_type = "MaxPool"; - } else { - auto iter = op_mapper_.find(pooling_type_); - onnx_pool_type = iter->second[0]; - } - auto node = helper_->MakeNode(onnx_pool_type, {input_x}); - helper_->AutoCast(node->output(0), output_info[0].name, P2ODataType::FP32, - output_info[0].dtype); - - AddAttribute(node, "kernel_shape", k_size_); - AddAttribute(node, "strides", strides_); - std::string auto_pad = "NOTSET"; - if (padding_algorithm_ == "SAME") { - auto_pad = "SAME_UPPER"; - AddAttribute(node, "auto_pad", auto_pad); - } else if (padding_algorithm_ == "VALID") { - auto_pad = "VALID"; - AddAttribute(node, "auto_pad", auto_pad); - } else { - AddAttribute(node, "pads", pads_); - } - if (OpType() != "max_pool2d_with_index" && helper_->GetOpsetVersion() >= 10) { - AddAttribute(node, "ceil_mode", static_cast(ceil_mode_)); - } - if (OpType() != "max_pool2d_with_index" && pooling_type_ == "avg") { - AddAttribute(node, "count_include_pad", static_cast(exclusive_)); - } -} - -int32_t Pool2dMapper::GetMinOpset(bool verbose) { - // NHWC is not supported - if (data_format_ == "NHWC") { - Error() << "NHWC format is not supported." << std::endl; - return -1; - } - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - if (IsAttrVar("ksize")) { - Error() << "While Attribute(ksize)'s type is Tensor, it's not " - "supported." - << std::endl; - return -1; - } else { - GetAttr("ksize", &k_size_); - } - - if (global_pooling_ || (k_size_[0] == 1 && k_size_[1] == 1)) { - if (ceil_mode_) { - Logger(verbose, 10) << "While ceil_model is True, " << RequireOpset(10) - << std::endl; - return 10; - } - return 7; - } - - if (adaptive_) { - for (auto one_input : input_info) { - for (auto i = 2; i < one_input.shape.size(); ++i) { - if (one_input.shape[i] == -1) { - if (export_as_custom_op) { - return 7; - } else { - Error() << "Adaptive only support static input shape." << std::endl; - return -1; - } - } - } - } - int64_t input_h = input_info[0].shape[2]; - int64_t input_w = input_info[0].shape[3]; - int64_t output_h = output_info[0].shape[2]; - int64_t output_w = output_info[0].shape[3]; - if (output_h == -1 || output_w == -1 || !IsSameSpan(input_h, output_h) || - !IsSameSpan(input_w, output_w)) { - if (export_as_custom_op) { - return 7; - } else { - Error() << "Cannot convert adaptive pool with input_size: " << input_h - << " " << input_h << " output_size: " << output_h << " " - << output_w << std::endl; - return -1; - } - } - } - if (OpType() == "max_pool2d_with_index") { - return 9; - } - auto iter = op_mapper_.find(pooling_type_); - if (op_mapper_.end() == iter) { - Error() << "Cannot find " << pooling_type_ << " in pool op_mapper." - << std::endl; - return -1; - } - - if (ceil_mode_) { - Logger(verbose, 10) << "While ceil_model is True, " << RequireOpset(10) - << std::endl; - return 10; - } - return 7; -} - -void Pool2dMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - GetAttr("ksize", &k_size_); - - bool is_1x1_kernel = true; - for (auto i : k_size_) { - if (i != 1) { - is_1x1_kernel = false; - } - } - - if (global_pooling_ || (adaptive_ && is_1x1_kernel)) { - std::string onnx_pool_type; - if (OpType() == "max_pool2d_with_index") { - onnx_pool_type = "GlobalMaxPool"; - } else { - auto iter = op_mapper_.find(pooling_type_); - onnx_pool_type = iter->second[1]; - } - auto input = helper_->AutoCast(input_info[0].name, input_info[0].dtype, - P2ODataType::FP32); - auto output = helper_->MakeNode(onnx_pool_type, {input})->output(0); - helper_->AutoCast(output, output_info[0].name, P2ODataType::FP32, - output_info[0].dtype); - } else if (adaptive_) { - AdaptivePool(input_info, output_info); - } else { - NoAdaptivePool(input_info, output_info); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/pool2d.h b/paddle2onnx/mapper/nn/pool2d.h deleted file mode 100644 index bbaf7c7879c..00000000000 --- a/paddle2onnx/mapper/nn/pool2d.h +++ /dev/null @@ -1,67 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class Pool2dMapper : public Mapper { - public: - Pool2dMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - op_mapper_["max"] = {"MaxPool", "GlobalMaxPool"}; - op_mapper_["avg"] = {"AveragePool", "GlobalAveragePool"}; - GetAttr("global_pooling", &global_pooling_); - GetAttr("adaptive", &adaptive_); - GetAttr("strides", &strides_); - GetAttr("paddings", &pads_); - if (OpType() != "max_pool2d_with_index") { - GetAttr("pooling_type", &pooling_type_); - GetAttr("data_format", &data_format_); - GetAttr("ceil_mode", &ceil_mode_); - GetAttr("padding_algorithm", &padding_algorithm_); - GetAttr("exclusive", &exclusive_); - exclusive_ = !exclusive_; - } - } - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - void ExportAsCustomOp(); - bool IsExportAsCustomOp(); - - private: - bool IsSameSpan(const int64_t& in_size, const int64_t& out_size); - void AdaptivePool(const std::vector& input_info, - const std::vector& output_info); - void NoAdaptivePool(const std::vector& input_info, - const std::vector& output_info); - bool ceil_mode_; - bool global_pooling_; - bool adaptive_; - bool exclusive_; - std::string data_format_; - std::string pooling_type_; - std::string padding_algorithm_; - std::vector k_size_; - std::vector pads_; - std::vector strides_; - std::map> op_mapper_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/pool3d.cc b/paddle2onnx/mapper/nn/pool3d.cc deleted file mode 100644 index fb6916fa100..00000000000 --- a/paddle2onnx/mapper/nn/pool3d.cc +++ /dev/null @@ -1,262 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/nn/pool3d.h" - -#include -#include -#include -#include - -namespace paddle2onnx { -REGISTER_MAPPER(pool3d, Pool3dMapper) -REGISTER_MAPPER(max_pool3d_with_index, Pool3dMapper) - -bool Pool3dMapper::IsSameSpan(const int64_t& in_size, const int64_t& out_size) { - std::vector spans; - spans.reserve(out_size); - for (auto i = 0; i < out_size; ++i) { - int64_t start = std::floor(i * (in_size / out_size)); - int64_t end = std::ceil((i + 1) * (in_size / out_size)); - spans.push_back(end - start); - } - std::sort(spans.begin(), spans.end()); - return spans[0] == spans[spans.size() - 1]; -} - -void Pool3dMapper::AdaptivePool(const std::vector& input_info, - const std::vector& output_info) { - int64_t input_d = input_info[0].shape[2]; - int64_t input_h = input_info[0].shape[3]; - int64_t input_w = input_info[0].shape[4]; - int64_t output_d = output_info[0].shape[2]; - int64_t output_h = output_info[0].shape[3]; - int64_t output_w = output_info[0].shape[4]; - int64_t stride_d = std::floor(input_d / output_d); - int64_t stride_h = std::floor(input_h / output_h); - int64_t stride_w = std::floor(input_w / output_w); - int64_t kernel_d = input_d - (output_d - 1) * stride_d; - int64_t kernel_h = input_h - (output_h - 1) * stride_h; - int64_t kernel_w = input_w - (output_w - 1) * stride_w; - std::string onnx_pool_type; - if (OpType() == "max_pool3d_with_index") { - onnx_pool_type = "MaxPool"; - } else { - auto iter = op_mapper_.find(pooling_type_); - onnx_pool_type = iter->second[0]; - } - std::shared_ptr* node_ptr; - auto input = helper_->AutoCast(input_info[0].name, input_info[0].dtype, - P2ODataType::FP32); - auto node = helper_->MakeNode(onnx_pool_type, {input}); - helper_->AutoCast(node->output(0), output_info[0].name, P2ODataType::FP32, - output_info[0].dtype); - std::vector kernel_size = {kernel_d, kernel_h, kernel_w}; - AddAttribute(node, "kernel_shape", kernel_size); - std::vector strides = {stride_d, stride_h, stride_w}; - AddAttribute(node, "strides", strides); - - if (helper_->GetOpsetVersion() > 10) { - AddAttribute(node, "ceil_mode", static_cast(ceil_mode_)); - } - - std::string auto_pad = "NOTSET"; - if (padding_algorithm_ == "SAME") { - auto_pad = "SAME_UPPER"; - } else if (padding_algorithm_ == "VALID") { - auto_pad = "VALID"; - } - AddAttribute(node, "auto_pad", auto_pad); - if (pooling_type_ == "avg") { - AddAttribute(node, "count_include_pad", static_cast(exclusive_)); - } -} - -void Pool3dMapper::NoAdaptivePool(const std::vector& input_info, - const std::vector& output_info) { - std::vector input_shape = input_info[0].shape; - if (pads_.size() == 3) { - pads_.push_back(pads_[0]); - pads_.push_back(pads_[1]); - pads_.push_back(pads_[2]); - } else if (pads_.size() == 6) { - std::vector index = {0, 2, 4, 1, 3, 5}; - std::vector copy = pads_; - for (auto i = 0; i < index.size(); ++i) { - pads_[i] = copy[index[i]]; - } - } - if (input_shape[2] > 0 && input_shape[2] + pads_[0] < k_size_[0]) { - k_size_[0] = input_shape[2] + pads_[0]; - } - if (input_shape[3] > 0 && input_shape[3] + pads_[1] < k_size_[1]) { - k_size_[1] = input_shape[3] + pads_[1]; - } - if (input_shape[4] > 0 && input_shape[4] + pads_[2] < k_size_[2]) { - k_size_[2] = input_shape[4] + pads_[2]; - } - - int64_t max_ksize = *std::max_element(std::begin(k_size_), std::end(k_size_)); - int64_t max_pads = *std::max_element(std::begin(pads_), std::end(pads_)); - auto input_x = helper_->AutoCast(input_info[0].name, input_info[0].dtype, - P2ODataType::FP32); - if (max_ksize <= max_pads) { - std::vector onnx_paddings = {0, 0, pads_[0], pads_[1], pads_[2], - 0, 0, pads_[3], pads_[4], pads_[5]}; - std::vector inputs_names = {input_x}; - if (helper_->GetOpsetVersion() >= 11) { - std::string paddings_node = - helper_->Constant(GetOnnxDtype(P2ODataType::INT64), onnx_paddings); - inputs_names.push_back(paddings_node); - std::vector val = {0.0}; - std::string val_node = - helper_->Constant(GetOnnxDtype(P2ODataType::FP32), val); - inputs_names.push_back(val_node); - } - auto node = helper_->MakeNode("Pad", inputs_names); - std::string mode = "constant"; - AddAttribute(node, "mode", mode); - if (helper_->GetOpsetVersion() < 11) { - AddAttribute(node, "pads", onnx_paddings); - float val = 0.0; - AddAttribute(node, "value", val); - } - input_x = node->output(0); - pads_.clear(); - pads_.resize(6, 0); - } - std::string onnx_pool_type; - if (OpType() == "max_pool3d_with_index") { - onnx_pool_type = "MaxPool"; - } else { - auto iter = op_mapper_.find(pooling_type_); - onnx_pool_type = iter->second[0]; - } - auto node = helper_->MakeNode(onnx_pool_type, {input_x}); - helper_->AutoCast(node->output(0), output_info[0].name, P2ODataType::FP32, - output_info[0].dtype); - - AddAttribute(node, "kernel_shape", k_size_); - AddAttribute(node, "strides", strides_); - std::string auto_pad = "NOTSET"; - if (padding_algorithm_ == "SAME") { - auto_pad = "SAME_UPPER"; - AddAttribute(node, "auto_pad", auto_pad); - } else if (padding_algorithm_ == "VALID") { - auto_pad = "VALID"; - AddAttribute(node, "auto_pad", auto_pad); - } else { - AddAttribute(node, "pads", pads_); - } - if (OpType() != "max_pool3d_with_index" && helper_->GetOpsetVersion() >= 10) { - AddAttribute(node, "ceil_mode", static_cast(ceil_mode_)); - } - if (pooling_type_ == "avg") { - AddAttribute(node, "count_include_pad", static_cast(exclusive_)); - } -} - -int32_t Pool3dMapper::GetMinOpset(bool verbose) { - // NHWC is not supported - if (data_format_ == "NDHWC") { - Error() << "NDHWC format is not supported." << std::endl; - return -1; - } - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - if (global_pooling_ || - (k_size_[0] == 1 && k_size_[1] == 1 && k_size_[2] == 1)) { - if (ceil_mode_) { - Logger(verbose, 10) << "While ceil_model is True, " << RequireOpset(10) - << std::endl; - return 10; - } - return 7; - } - - if (adaptive_) { - for (auto one_input : input_info) { - for (auto i = 2; i < one_input.shape.size(); ++i) { - if (one_input.shape[i] == -1) { - Error() << "Adaptive only support static input shape." << std::endl; - return -1; - } - } - } - int64_t input_d = input_info[0].shape[2]; - int64_t input_h = input_info[0].shape[3]; - int64_t input_w = input_info[0].shape[4]; - int64_t output_d = output_info[0].shape[2]; - int64_t output_h = output_info[0].shape[3]; - int64_t output_w = output_info[0].shape[4]; - if (!IsSameSpan(input_h, output_h) || !IsSameSpan(input_w, output_w) || - !IsSameSpan(input_d, output_d)) { - Error() << "Cannot convert adaptive pool with input_size: " << input_d - << " " << input_h << " " << input_w - << " output_size: " << output_d << " " << output_h << " " - << output_w << std::endl; - return -1; - } - } - if (OpType() == "max_pool3d_with_index") { - return 9; - } - auto iter = op_mapper_.find(pooling_type_); - if (op_mapper_.end() == iter) { - Error() << "Cannot find " << pooling_type_ << " in pool op_mapper." - << std::endl; - return -1; - } - - if (ceil_mode_) { - Logger(verbose, 10) << "While ceil_model is True, " << RequireOpset(10) - << std::endl; - return 10; - } - return 7; -} - -void Pool3dMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - bool is_1x1_kernel = true; - for (auto i : k_size_) { - if (i != 1) { - is_1x1_kernel = false; - } - } - - if (global_pooling_ || (adaptive_ && is_1x1_kernel)) { - std::string onnx_pool_type; - if (OpType() == "max_pool3d_with_index") { - onnx_pool_type = "GlobalMaxPool"; - } else { - auto iter = op_mapper_.find(pooling_type_); - onnx_pool_type = iter->second[1]; - } - auto input = helper_->AutoCast(input_info[0].name, input_info[0].dtype, - P2ODataType::FP32); - auto output = helper_->MakeNode(onnx_pool_type, {input})->output(0); - helper_->AutoCast(output, output_info[0].name, P2ODataType::FP32, - output_info[0].dtype); - } else if (adaptive_) { - AdaptivePool(input_info, output_info); - } else { - NoAdaptivePool(input_info, output_info); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/pool3d.h b/paddle2onnx/mapper/nn/pool3d.h deleted file mode 100644 index 9be0c901df9..00000000000 --- a/paddle2onnx/mapper/nn/pool3d.h +++ /dev/null @@ -1,66 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class Pool3dMapper : public Mapper { - public: - Pool3dMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - op_mapper_["max"] = {"MaxPool", "GlobalMaxPool"}; - op_mapper_["avg"] = {"AveragePool", "GlobalAveragePool"}; - GetAttr("global_pooling", &global_pooling_); - GetAttr("adaptive", &adaptive_); - GetAttr("strides", &strides_); - GetAttr("paddings", &pads_); - GetAttr("ksize", &k_size_); - if (OpType() != "max_pool3d_with_index") { - GetAttr("pooling_type", &pooling_type_); - GetAttr("data_format", &data_format_); - GetAttr("ceil_mode", &ceil_mode_); - GetAttr("padding_algorithm", &padding_algorithm_); - GetAttr("exclusive", &exclusive_); - exclusive_ = !exclusive_; - } - } - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - bool IsSameSpan(const int64_t& in_size, const int64_t& out_size); - void AdaptivePool(const std::vector& input_info, - const std::vector& output_info); - void NoAdaptivePool(const std::vector& input_info, - const std::vector& output_info); - bool ceil_mode_; - bool global_pooling_; - bool adaptive_; - bool exclusive_; - std::string data_format_; - std::string pooling_type_; - std::string padding_algorithm_; - std::vector k_size_; - std::vector pads_; - std::vector strides_; - std::map> op_mapper_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/rnn.cc b/paddle2onnx/mapper/nn/rnn.cc deleted file mode 100644 index 1c6f609fb06..00000000000 --- a/paddle2onnx/mapper/nn/rnn.cc +++ /dev/null @@ -1,156 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/nn/rnn.h" -#include - -namespace paddle2onnx { -REGISTER_MAPPER(rnn, RnnMapper) - -int32_t RnnMapper::GetMinOpset(bool verbose) { - return 7; -} - -std::string RnnMapper::ReformWeight(const std::string& weight, const int64_t& size, const std::vector& perm) { - std::vector items; - for (size_t i = 0; i < perm.size(); i += 2) { - auto item = helper_->Slice(weight, {1}, {perm[i] * size}, {perm[i + 1] * size}); - items.push_back(item); - } - return helper_->Concat(items, 1); -} - -std::vector RnnMapper::MakeParamInputs(int64_t layer_index) { - auto weight_list_info = GetInput("WeightList"); - int64_t bidirect_len = is_bidirec_ ? 4 : 2; - int64_t all_layer_param_len = weight_list_info.size(); - int64_t single_layer_param_len = std::floor(all_layer_param_len / num_layers_); - int64_t weight_start_idx = layer_index * bidirect_len; - int64_t weight_end_idx = weight_start_idx + bidirect_len; - int64_t bias_start_idx = weight_start_idx + std::floor(all_layer_param_len / 2); - int64_t bias_end_idx = bias_start_idx + bidirect_len; - - std::vector unsqueezed_weights; - for (auto i = weight_start_idx; i < weight_end_idx; ++i) { - unsqueezed_weights.push_back(helper_->Unsqueeze(weight_list_info[i].name, {0})); - } - for (auto i = bias_start_idx; i < bias_end_idx; ++i) { - unsqueezed_weights.push_back(helper_->Unsqueeze(weight_list_info[i].name, {0})); - } - - std::vector input_weight; - std::vector hidden_weight; - for (size_t i = 0; i < bidirect_len; i += 2) { - input_weight.push_back(unsqueezed_weights[i]); - } - for (size_t i = 1; i < bidirect_len; i += 2) { - hidden_weight.push_back(unsqueezed_weights[i]); - } - std::vector input_bias; - std::vector hidden_bias; - for (size_t i = bidirect_len; i < 2 * bidirect_len; i += 2) { - input_bias.push_back(unsqueezed_weights[i]); - } - for (size_t i = bidirect_len + 1; i < 2 * bidirect_len; i += 2) { - hidden_bias.push_back(unsqueezed_weights[i]); - } - - auto input_weight_tensor = helper_->Concat(input_weight, 0); - auto hidden_weight_tensor = helper_->Concat(hidden_weight, 0); - auto input_bias_tensor = helper_->Concat(input_bias, 0); - auto hidden_bias_tensor = helper_->Concat(hidden_bias, 0); - - std::vector reform_permutation; - if (mode_ == "LSTM") { - std::vector perm({0, 1, 3, 4, 1, 3}); - reform_permutation.assign(perm.begin(), perm.end()); - } else if (mode_ == "GRU") { - std::vector perm({1, 2, 0, 1, 2, 3}); - reform_permutation.assign(perm.begin(), perm.end()); - } - input_weight_tensor = ReformWeight(input_weight_tensor, hidden_size_, reform_permutation); - hidden_weight_tensor = ReformWeight(hidden_weight_tensor, hidden_size_, reform_permutation); - input_bias_tensor = ReformWeight(input_bias_tensor, hidden_size_, reform_permutation); - hidden_bias_tensor = ReformWeight(hidden_bias_tensor, hidden_size_, reform_permutation); - - std::vector outputs; - outputs.push_back(input_weight_tensor); - outputs.push_back(hidden_weight_tensor); - outputs.push_back(helper_->Concat({input_bias_tensor, hidden_bias_tensor}, 1)); - outputs.push_back(""); - return outputs; -} - -std::vector RnnMapper::MakeInitParamInputs(int64_t layer_index) { - std::vector outputs; - auto prestate_info = GetInput("PreState"); - int64_t bidirect_len = is_bidirec_ ? 2 : 1; - auto init_h = helper_->Slice(prestate_info[0].name, {0}, {layer_index * bidirect_len}, {layer_index * bidirect_len + bidirect_len}); - outputs.push_back(init_h); - if (mode_ == "GRU") { - return outputs; - } - auto init_c = helper_->Slice(prestate_info[1].name, {0}, {layer_index * bidirect_len}, {layer_index * bidirect_len + bidirect_len}); - outputs.push_back(init_c); - return outputs; -} - -void RnnMapper::Opset7() { - auto input_info = GetInput("Input"); - auto state_info = GetOutput("State"); - auto out_info = GetOutput("Out"); - auto input = input_info[0].name; - if (mode_ == "LSTM") { - std::string h_out = ""; - std::string c_out = ""; - for (auto i = 0; i < num_layers_; ++i) { - auto param_inputs = MakeParamInputs(i); - auto init_param_inputs = MakeInitParamInputs(i); - std::vector inputs({input}); - inputs.insert(inputs.end(), param_inputs.begin(), param_inputs.end()); - inputs.insert(inputs.end(), init_param_inputs.begin(), init_param_inputs.end()); - auto node = helper_->MakeNode("LSTM", inputs, 3); - std::string direction = is_bidirec_ ? "bidirectional" : "forward"; - AddAttribute(node, "direction", direction); - AddAttribute(node, "hidden_size", hidden_size_); - input = helper_->Transpose(node->output(0), {0, 2, 1, 3}); - input = helper_->Reshape(input, {0, 0, -1}); - h_out = node->output(1); - c_out = node->output(2); - } - helper_->MakeNode("Identity", {h_out}, {state_info[0].name}); - helper_->MakeNode("Identity", {c_out}, {state_info[1].name}); - helper_->MakeNode("Identity", {input}, {out_info[0].name}); - } else if (mode_ == "GRU") { - std::string h_out = ""; - for (auto i = 0; i < num_layers_; ++i) { - auto param_inputs = MakeParamInputs(i); - auto init_param_inputs = MakeInitParamInputs(i); - std::vector inputs({input}); - inputs.insert(inputs.end(), param_inputs.begin(), param_inputs.end()); - inputs.insert(inputs.end(), init_param_inputs.begin(), init_param_inputs.end()); - auto node = helper_->MakeNode("GRU", inputs, 2); - std::string direction = is_bidirec_ ? "bidirectional" : "forward"; - AddAttribute(node, "direction", direction); - AddAttribute(node, "hidden_size", hidden_size_); - AddAttribute(node, "linear_before_reset", int64_t(1)); - input = helper_->Transpose(node->output(0), {0, 2, 1, 3}); - input = helper_->Reshape(input, {0, 0, -1}); - h_out = node->output(1); - } - helper_->MakeNode("Identity", {h_out}, {state_info[0].name}); - helper_->MakeNode("Identity", {input}, {out_info[0].name}); - } -} -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/rnn.h b/paddle2onnx/mapper/nn/rnn.h deleted file mode 100644 index b2ef24df1f5..00000000000 --- a/paddle2onnx/mapper/nn/rnn.h +++ /dev/null @@ -1,58 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class RnnMapper : public Mapper { - public: - RnnMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - MarkAsExperimentalOp(); - GetAttr("num_layers", &num_layers_); - GetAttr("input_size", &input_size_); - GetAttr("hidden_size", &hidden_size_); - GetAttr("seed", &seed_); - GetAttr("dropout_prob", &dropout_prob_); - GetAttr("mode", &mode_); - GetAttr("is_bidirec", &is_bidirec_); - if (HasAttr("is_test")) { - GetAttr("is_test", &is_test_); - } - } - - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - std::vector MakeParamInputs(int64_t layer_index); - std::vector MakeInitParamInputs(int64_t layer_index); - std::string ReformWeight(const std::string& weight, const int64_t& size, const std::vector& perm); - int64_t num_layers_; - int64_t input_size_; - int64_t hidden_size_; - int64_t seed_; - float dropout_prob_; - std::string mode_; - bool is_test_ = false; - bool is_bidirec_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/shape.cc b/paddle2onnx/mapper/nn/shape.cc deleted file mode 100644 index d82a1d00f4f..00000000000 --- a/paddle2onnx/mapper/nn/shape.cc +++ /dev/null @@ -1,29 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/nn/shape.h" - -namespace paddle2onnx { -REGISTER_MAPPER(shape, ShapeMapper) - -void ShapeMapper::Opset7() { - auto input_info = GetInput("Input"); - auto output_info = GetOutput("Out"); - - auto shape_out = helper_->MakeNode("Shape", {input_info[0].name})->output(0); - helper_->AutoCast(shape_out, output_info[0].name, P2ODataType::INT64, - output_info[0].dtype); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/shape.h b/paddle2onnx/mapper/nn/shape.h deleted file mode 100644 index 6fd1867c66d..00000000000 --- a/paddle2onnx/mapper/nn/shape.h +++ /dev/null @@ -1,32 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class ShapeMapper : public Mapper { - public: - ShapeMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - - void Opset7(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/softmax_with_cross_entropy.cc b/paddle2onnx/mapper/nn/softmax_with_cross_entropy.cc deleted file mode 100644 index 14c64e9d07c..00000000000 --- a/paddle2onnx/mapper/nn/softmax_with_cross_entropy.cc +++ /dev/null @@ -1,126 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/nn/softmax_with_cross_entropy.h" - -namespace paddle2onnx { -REGISTER_MAPPER(softmax_with_cross_entropy, SoftmaxCrossEntropyLossMapper) - -int32_t SoftmaxCrossEntropyLossMapper::GetMinOpset(bool verbose) { - auto logits = GetInput("Logits"); - std::vector logits_shape = logits[0].shape; - if (logits_shape.size() < 2) { - Error() << "SoftmaxCrossEntropyLoss in onnx not support 1D logits." - << std::endl; - return -1; - } - Logger(verbose, 12) << RequireOpset(12) << std::endl; - return 12; -} - -void SoftmaxCrossEntropyLossMapper::Opset12() { - auto logits = GetInput("Logits"); - auto labels = GetInput("Label"); - - auto loss = GetOutput("Loss"); - auto softmax = GetOutput("Softmax"); - std::vector logits_shape = logits[0].shape; - auto dim = logits[0].Rank(); - if (axis_ < 0) { - axis_ += dim; - } - if (soft_label_) { - std::vector split; - split.resize(logits_shape[axis_], 1); - std::vector axes_val = {axis_}; - std::string axes_node = - helper_->Constant(GetOnnxDtype(P2ODataType::INT64), axes_val); - if (axis_ == dim - 1) { - auto logsoftmax_node = helper_->MakeNode("LogSoftmax", {logits[0].name}); - AddAttribute(logsoftmax_node, "axis", axis_); - helper_->MakeNode("Exp", {logsoftmax_node->output(0)}, {softmax[0].name}); - auto mul_result = helper_->MakeNode( - "Mul", {logsoftmax_node->output(0), labels[0].name}); - if (helper_->GetOpsetVersion() < 13) { - auto reducesum_node = - helper_->MakeNode("ReduceSum", {mul_result->output(0)}); - AddAttribute(reducesum_node, "axes", axes_val); - helper_->MakeNode("Neg", {reducesum_node->output(0)}, {loss[0].name}); - } else { - auto reducesum_node = - helper_->MakeNode("ReduceSum", {mul_result->output(0), axes_node}); - helper_->MakeNode("Neg", {reducesum_node->output(0)}, {loss[0].name}); - } - } else { - auto perm = Arange(0, dim); - perm[dim - 1] = axis_; - perm[axis_] = dim - 1; - auto output = helper_->Transpose(logits[0].name, perm); - auto logsoftmax_node = helper_->MakeNode("LogSoftmax", {output}); - AddAttribute(logsoftmax_node, "axis", int64_t(-1)); - auto transpose_logsoftmax_node = - helper_->Transpose(logsoftmax_node->output(0), perm); - helper_->MakeNode("Exp", {transpose_logsoftmax_node}, {softmax[0].name}); - auto mul_result = - helper_->MakeNode("Mul", {transpose_logsoftmax_node, labels[0].name}); - if (helper_->GetOpsetVersion() < 13) { - auto reducesum_node = - helper_->MakeNode("ReduceSum", {mul_result->output(0)}); - AddAttribute(reducesum_node, "axes", axes_val); - helper_->MakeNode("Neg", {reducesum_node->output(0)}, {loss[0].name}); - } else { - auto reducesum_node = - helper_->MakeNode("ReduceSum", {mul_result->output(0), axes_node}); - helper_->MakeNode("Neg", {reducesum_node->output(0)}, {loss[0].name}); - } - } - } else { - if (axis_ == 1) { - auto squeeze_node = helper_->Squeeze(labels[0].name, {axis_}); - auto node = helper_->MakeNode("SoftmaxCrossEntropyLoss", - {logits[0].name, squeeze_node}, 2); - AddAttribute(node, "ignore_index", ignore_index_); - AddAttribute(node, "reduction", "none"); - auto loss_node = - helper_->Unsqueeze(node->output(0), loss[0].name, {axis_}); - // onnx output is log(softmax), but paddle output is softmax - helper_->MakeNode("Exp", {node->output(1)}, {softmax[0].name}); - } else { - std::vector perm = Arange(0, dim); - perm[1] = axis_; - perm[axis_] = 1; - auto transpose_logits = helper_->MakeNode("Transpose", {logits[0].name}); - AddAttribute(transpose_logits, "perm", perm); - auto transpose_labels = helper_->MakeNode("Transpose", {labels[0].name}); - AddAttribute(transpose_labels, "perm", perm); - auto squeeze_labels = helper_->Squeeze(transpose_labels->output(0), {1}); - auto node = - helper_->MakeNode("SoftmaxCrossEntropyLoss", - {transpose_logits->output(0), squeeze_labels}, 2); - AddAttribute(node, "ignore_index", ignore_index_); - AddAttribute(node, "reduction", "none"); - auto unsqueeze_node = helper_->Unsqueeze(node->output(0), {1}); - auto revert_transpose_logits = - helper_->MakeNode("Transpose", {unsqueeze_node}, {loss[0].name}); - AddAttribute(revert_transpose_logits, "perm", perm); - auto revert_transpose_softmax = - helper_->MakeNode("Transpose", {node->output(1)}); - AddAttribute(revert_transpose_softmax, "perm", perm); - // onnx output is log(softmax), but paddle output is softmax - helper_->MakeNode("Exp", {revert_transpose_softmax->output(0)}, - {softmax[0].name}); - } - } -} -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/nn/softmax_with_cross_entropy.h b/paddle2onnx/mapper/nn/softmax_with_cross_entropy.h deleted file mode 100644 index 960f978de64..00000000000 --- a/paddle2onnx/mapper/nn/softmax_with_cross_entropy.h +++ /dev/null @@ -1,40 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class SoftmaxCrossEntropyLossMapper : public Mapper { - public: - SoftmaxCrossEntropyLossMapper(const PaddleParser& p, OnnxHelper* helper, - int64_t block_id, int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("axis", &axis_); - GetAttr("soft_label", &soft_label_); - GetAttr("ignore_index", &ignore_index_); - } - int32_t GetMinOpset(bool verbose = false); - void Opset12(); - - private: - int64_t axis_ = -1; - bool soft_label_ = false; - int64_t ignore_index_ = -100; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/onnx_helper.cc b/paddle2onnx/mapper/onnx_helper.cc deleted file mode 100755 index 2e37005af5d..00000000000 --- a/paddle2onnx/mapper/onnx_helper.cc +++ /dev/null @@ -1,534 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/onnx_helper.h" - -#include - -namespace paddle2onnx { - -void AddAttribute(std::shared_ptr node, - const std::string& name, const int64_t& value) { - for (int i = 0; i < node->attribute_size(); ++i) { - if (node->attribute(i).name() == name) { - node->mutable_attribute(i)->set_i(value); - node->mutable_attribute(i)->set_type(ONNX_NAMESPACE::AttributeProto::INT); - return; - } - } - auto attr = node->add_attribute(); - attr->set_name(name); - attr->set_i(value); - attr->set_type(ONNX_NAMESPACE::AttributeProto::INT); -} - -void AddAttribute(std::shared_ptr node, - const std::string& name, const float& value) { - for (int i = 0; i < node->attribute_size(); ++i) { - if (node->attribute(i).name() == name) { - node->mutable_attribute(i)->set_f(value); - node->mutable_attribute(i)->set_type( - ONNX_NAMESPACE::AttributeProto::FLOAT); - return; - } - } - auto attr = node->add_attribute(); - attr->set_name(name); - attr->set_f(value); - attr->set_type(ONNX_NAMESPACE::AttributeProto::FLOAT); -} - -void AddAttribute(std::shared_ptr node, - const std::string& name, const std::string& value) { - auto attr = node->add_attribute(); - attr->set_name(name); - attr->set_s(value); - attr->set_type(ONNX_NAMESPACE::AttributeProto::STRING); -} - -void AddAttribute(std::shared_ptr node, - const std::string& name, const std::vector& values) { - auto attr = node->add_attribute(); - attr->set_name(name); - for (auto& item : values) { - attr->add_ints(item); - } - attr->set_type(ONNX_NAMESPACE::AttributeProto::INTS); -} - -void AddAttribute(std::shared_ptr node, - const std::string& name, const std::vector& values) { - auto attr = node->add_attribute(); - attr->set_name(name); - for (auto& item : values) { - attr->add_floats(item); - } - attr->set_type(ONNX_NAMESPACE::AttributeProto::FLOATS); -} - -void AddAttribute(std::shared_ptr node, - const std::string& name, - ONNX_NAMESPACE::TensorProto_DataType dtype) { - auto attr = node->add_attribute(); - attr->set_name(name); - attr->set_i(static_cast(dtype)); - attr->set_type(ONNX_NAMESPACE::AttributeProto::INT); -} - -ONNX_NAMESPACE::TensorProto_DataType GetOnnxDtype(int32_t paddle_dtype) { - Assert((paddle_dtype >= 0 && paddle_dtype <= 6) || paddle_dtype == 20 || - paddle_dtype == 21, - "Unknow paddle data type: " + std::to_string(paddle_dtype) + - " While call GetOnnxDtype."); - auto onnx_dtype = ONNX_NAMESPACE::TensorProto::FLOAT; - if (paddle_dtype == P2ODataType::BOOL) { - onnx_dtype = ONNX_NAMESPACE::TensorProto::BOOL; - } else if (paddle_dtype == P2ODataType::INT8) { - onnx_dtype = ONNX_NAMESPACE::TensorProto::INT8; - } else if (paddle_dtype == P2ODataType::INT16) { - onnx_dtype = ONNX_NAMESPACE::TensorProto::INT16; - } else if (paddle_dtype == P2ODataType::INT32) { - onnx_dtype = ONNX_NAMESPACE::TensorProto::INT32; - } else if (paddle_dtype == P2ODataType::INT64) { - onnx_dtype = ONNX_NAMESPACE::TensorProto::INT64; - } else if (paddle_dtype == P2ODataType::FP16) { - onnx_dtype = ONNX_NAMESPACE::TensorProto::FLOAT16; - } else if (paddle_dtype == P2ODataType::FP32) { - onnx_dtype = ONNX_NAMESPACE::TensorProto::FLOAT; - } else if (paddle_dtype == P2ODataType::FP64) { - onnx_dtype = ONNX_NAMESPACE::TensorProto::DOUBLE; - } else { - onnx_dtype = ONNX_NAMESPACE::TensorProto::UINT8; - } - return onnx_dtype; -} - -std::shared_ptr MakeConstant(const std::string& name, - const Weight& weight) { - auto node = std::make_shared(); - node->set_op_type("Constant"); - node->add_output(name); - auto attr = node->add_attribute(); - attr->set_name("value"); - attr->set_type(ONNX_NAMESPACE::AttributeProto::TENSOR); - auto tensor = attr->mutable_t(); - tensor->set_name(name); - auto onnx_dtype = GetOnnxDtype(weight.dtype); - tensor->set_data_type(onnx_dtype); - for (auto& dim : weight.shape) { - tensor->add_dims(dim); - } - tensor->set_raw_data(std::string(weight.buffer.data(), weight.buffer.size())); - return node; -} - -// std::shared_ptr OnnxHelper::MakeConstant( -// const Weight& weight) { -// auto node_name = MapperHelper::Get()->GenName("auto.constant"); -// return MakeConstant(node_name, weight); -//} -// -// std::shared_ptr OnnxHelper::MakeConstant( -// const std::string& name, const Weight& weight) { -// auto node = std::make_shared(); -// node->set_op_type("Constant"); -// node->add_output(name); -// auto attr = node->add_attribute(); -// attr->set_name("value"); -// attr->set_type(ONNX_NAMESPACE::AttributeProto::TENSOR); -// auto tensor = attr->mutable_t(); -// tensor->set_name(name); -// auto onnx_dtype = GetOnnxDtype(weight.dtype); -// tensor->set_data_type(onnx_dtype); -// for (auto& dim : weight.shape) { -// tensor->add_dims(dim); -// } -// tensor->set_raw_data(std::string(weight.buffer.data(), -// weight.buffer.size())); -// nodes.push_back(node); -// return node; -//} - -std::shared_ptr MakeValueInfo( - const TensorInfo& info) { - auto value_info = std::make_shared(); - value_info->set_name(info.name); - auto type_proto = value_info->mutable_type(); - auto tensor_type_proto = type_proto->mutable_tensor_type(); - tensor_type_proto->set_elem_type(GetOnnxDtype(info.dtype)); - auto shape = tensor_type_proto->mutable_shape(); - for (auto& dim : info.shape) { - if (dim < 0) { - auto dynamic_dim_name = MapperHelper::Get()->GenName("DynamicDimension"); - shape->add_dim()->set_dim_param(dynamic_dim_name); - } else { - shape->add_dim()->set_dim_value(dim); - } - } - return value_info; -} - -std::shared_ptr OnnxHelper::MakeValueInfo( - const std::string& name, const int32_t& dtype, - std::vector& shape) { - auto value_info = std::make_shared(); - value_info->set_name(name); - auto type_proto = value_info->mutable_type(); - auto tensor_type_proto = type_proto->mutable_tensor_type(); - tensor_type_proto->set_elem_type(GetOnnxDtype(dtype)); - auto shape_proto = tensor_type_proto->mutable_shape(); - for (auto& dim : shape) { - if (dim < 0) { - auto dynamic_dim_name = MapperHelper::Get()->GenName("DynamicDimension"); - shape_proto->add_dim()->set_dim_param(dynamic_dim_name); - } else { - shape_proto->add_dim()->set_dim_value(dim); - } - } - value_infos.push_back(value_info); - return value_info; -} - -std::shared_ptr OnnxHelper::MakeNode( - const std::string& op_type, const std::vector& inputs, - const std::vector& outputs) { -#ifdef PADDLE2ONNX_DEBUG - P2OLogger(true) << "ONNX Node: " << op_type << std::endl; -#endif - auto node = std::make_shared(); - auto node_name = MapperHelper::Get()->GenName(op_type); - node->set_name(node_name); - node->set_op_type(op_type); - for (size_t i = 0; i < inputs.size(); ++i) { - node->add_input(inputs[i]); - } - for (size_t i = 0; i < outputs.size(); ++i) { - node->add_output(outputs[i]); - } - if (op_type == "Reshape" && GetOpsetVersion() >= 14) { - AddAttribute(node, "allowzero", int64_t(0)); - } - - nodes.push_back(node); - return node; -} - -std::shared_ptr OnnxHelper::MakeNode( - const std::string& op_type, const std::vector& inputs, - int num_outputs) { -#ifdef PADDLE2ONNX_DEBUG - P2OLogger(true) << "ONNX Node: " << op_type << std::endl; -#endif - auto node = std::make_shared(); - auto node_name = MapperHelper::Get()->GenName(op_type); - node->set_name(node_name); - node->set_op_type(op_type); - for (size_t i = 0; i < inputs.size(); ++i) { - node->add_input(inputs[i]); - } - std::vector outputs; - for (auto i = 0; i < num_outputs; ++i) { - outputs.push_back(MapperHelper::Get()->GenName(op_type)); - } - for (size_t i = 0; i < outputs.size(); ++i) { - node->add_output(outputs[i]); - } - if (op_type == "Reshape" && GetOpsetVersion() >= 14) { - AddAttribute(node, "allowzero", int64_t(0)); - } - nodes.push_back(node); - return node; -} - -std::string OnnxHelper::AutoCast(const std::string& input, - int32_t input_paddle_dtype, - int32_t to_paddle_dtype) { - std::string output = MapperHelper::Get()->GenName("auto.cast"); - if (input_paddle_dtype == to_paddle_dtype) { - MakeNode("Identity", {input}, {output}); - return output; - } - auto cast_node = MakeNode("Cast", {input}, {output}); - AddAttribute(cast_node, "to", GetOnnxDtype(to_paddle_dtype)); - return cast_node->output(0); -} - -std::string OnnxHelper::AutoCast(const std::string& input, - const std::string& output, - int32_t input_paddle_dtype, - int32_t to_paddle_dtype) { - if (input_paddle_dtype == to_paddle_dtype) { - auto node = MakeNode("Identity", {input}, {output}); - return output; - } - auto cast_node = MakeNode("Cast", {input}, {output}); - AddAttribute(cast_node, "to", GetOnnxDtype(to_paddle_dtype)); - return cast_node->output(0); -} - -std::string OnnxHelper::ConcatIndices(const std::vector& indices) { - std::vector vars; - // make sure all the indices be 1-D tensor - for (size_t i = 0; i < indices.size(); ++i) { - std::string var = indices[i].name; - if (indices[i].Rank() != 1) { - var = Reshape(indices[i].name, {1}); - } - vars.push_back(var); - } - // make sure all the indices be int64 - for (size_t i = 0; i < indices.size(); ++i) { - if (indices[i].dtype != P2ODataType::INT64) { - auto node = MakeNode("Cast", {vars[i]}); - AddAttribute(node, "to", ONNX_NAMESPACE::TensorProto::INT64); - vars[i] = node->output(0); - } - } - // concat and return - if (vars.size() > 1) { - return Concat(vars, 0); - } - return vars[0]; -} - -std::string OnnxHelper::Clip(const std::string& input, - const std::string& output, const float& min, - const float& max, const int32_t& in_dtype) { - // onnxruntime only supports float input - std::string input_name = AutoCast(input, in_dtype, P2ODataType::FP32); - if (opset_version < 11) { - auto node = MakeNode("Clip", {input_name}); - AddAttribute(node, "max", max); - AddAttribute(node, "min", min); - auto res = AutoCast(node->output(0), output, P2ODataType::FP32, in_dtype); - return res; - } else { - int32_t dtype = P2ODataType::FP32; - std::string min_name = Constant({}, GetOnnxDtype(dtype), min); - std::string max_name; - max_name = Constant({}, GetOnnxDtype(dtype), max); - auto node = MakeNode("Clip", {input_name, min_name, max_name}); - auto res = AutoCast(node->output(0), {output}, P2ODataType::FP32, in_dtype); - return res; - } -} - -std::string OnnxHelper::Clip(const std::string& input, const float& min, - const float& max, const int32_t& in_dtype) { - std::string output = MapperHelper::Get()->GenName("helper.clip"); - return Clip(input, output, min, max, in_dtype); -} - -std::string OnnxHelper::Squeeze(const std::string& input, - const std::string& output, - const std::vector& axes) { - if (axes.size() == 0) { - auto node = MakeNode("Squeeze", {input}, {output}); - } else { - if (opset_version < 13) { - auto node = MakeNode("Squeeze", {input}, {output}); - AddAttribute(node, "axes", axes); - } else { - auto axes_node = Constant(ONNX_NAMESPACE::TensorProto::INT64, axes); - auto node = MakeNode("Squeeze", {input, axes_node}, {output}); - } - } - return output; -} - -std::string OnnxHelper::Squeeze(const std::string& input, - const std::vector& axes) { - std::string output = MapperHelper::Get()->GenName("helper.squeeze"); - return Squeeze(input, output, axes); -} - -std::string OnnxHelper::Unsqueeze(const std::string& input, - const std::string& output, - const std::vector& axes) { - Assert(axes.size() >= 0, "OnnxHelper::Unsqueeze Size of axes should > 0"); - for (auto& item : axes) { - Assert(item >= 0, - "OnnxHelper::Unsqueeze All the elements in axes should >= 0"); - } - if (opset_version < 13) { - auto node = MakeNode("Unsqueeze", {input}, {output}); - AddAttribute(node, "axes", axes); - } else { - auto axes_node = Constant(ONNX_NAMESPACE::TensorProto::INT64, axes); - auto node = MakeNode("Unsqueeze", {input, axes_node}, {output}); - } - return output; -} - -std::string OnnxHelper::Unsqueeze(const std::string& input, - const std::vector& axes) { - std::string output = MapperHelper::Get()->GenName("helper.unsqueeze"); - return Unsqueeze(input, output, axes); -} - -std::string OnnxHelper::Reshape(const std::string& input, - const std::string& output, - const std::vector& shape) { - if (opset_version < 6) { - auto node = MakeNode("Reshape", {input}, {output}); - AddAttribute(node, "shape", shape); - } else { - auto shape_node = Constant(ONNX_NAMESPACE::TensorProto::INT64, shape); - auto node = MakeNode("Reshape", {input, shape_node}, {output}); - if (opset_version >= 14) { - AddAttribute(node, "allowzero", int64_t(0)); - } - } - return output; -} - -std::string OnnxHelper::Reshape(const std::string& input, - const std::vector& shape) { - std::string output = MapperHelper::Get()->GenName("helper.reshape"); - return Reshape(input, output, shape); -} - -std::string OnnxHelper::Flatten(const std::string& input, - const std::string& output) { - return Reshape(input, output, std::vector(1, -1)); -} - -std::string OnnxHelper::Flatten(const std::string& input) { - std::string output = MapperHelper::Get()->GenName("helper.flatten"); - return Flatten(input, output); -} - -std::string OnnxHelper::Slice(const std::string& input, - const std::string& output, - const std::vector& axes, - const std::vector& starts, - const std::vector& ends) { - if (opset_version < 10) { - auto node = MakeNode("Slice", {input}, {output}); - AddAttribute(node, "axes", axes); - AddAttribute(node, "starts", starts); - AddAttribute(node, "ends", ends); - } else { - auto axes_node = Constant(ONNX_NAMESPACE::TensorProto::INT64, axes); - auto starts_node = Constant(ONNX_NAMESPACE::TensorProto::INT64, starts); - auto ends_node = Constant(ONNX_NAMESPACE::TensorProto::INT64, ends); - auto node = - MakeNode("Slice", {input, starts_node, ends_node, axes_node}, {output}); - } - return output; -} - -std::string OnnxHelper::Slice(const std::string& input, - const std::vector& axes, - const std::vector& starts, - const std::vector& ends) { - std::string output = MapperHelper::Get()->GenName("helper.slice"); - return Slice(input, output, axes, starts, ends); -} - -std::string OnnxHelper::Concat(const std::vector& input, - const std::string& output, int64_t axis) { - auto node = MakeNode("Concat", input, {output}); - AddAttribute(node, "axis", axis); - return output; -} - -std::string OnnxHelper::Concat(const std::vector& input, - int64_t axis) { - auto output = MapperHelper::Get()->GenName("helper.concat"); - return Concat(input, output, axis); -} - -std::string OnnxHelper::Transpose(const std::string& input, - const std::string& output, - const std::vector& perm) { - auto node = MakeNode("Transpose", {input}, {output}); - AddAttribute(node, "perm", perm); - return output; -} - -std::string OnnxHelper::Transpose(const std::string& input, - const std::vector& perm) { - auto output = MapperHelper::Get()->GenName("helper.transpose"); - return Transpose(input, output, perm); -} - -std::vector OnnxHelper::Split( - const std::string& input, const std::vector& outputs, - const std::vector& split, int64_t axis) { - Assert(outputs.size() > 0 || split.size() > 0, - "OnnxHelper::Split requires the size of outputs or the size of split " - "> 0."); - auto node = std::make_shared(); - auto node_name = MapperHelper::Get()->GenName("Split"); - node->set_name(node_name); - node->set_op_type("Split"); - node->add_input(input); - for (size_t i = 0; i < outputs.size(); ++i) { - node->add_output(outputs[i]); - } - AddAttribute(node, "axis", axis); - if (split.size() > 0) { - Assert(outputs.size() == split.size(), - "OnnxHelper::Split While size of outputs and the size of split both " - "> 0, their size must be same."); - if (opset_version < 13) { - AddAttribute(node, "split", split); - } else { - auto split_const = Constant(ONNX_NAMESPACE::TensorProto::INT64, split); - node->add_input(split_const); - } - } - nodes.push_back(node); - return outputs; -} - -std::vector OnnxHelper::Split(const std::string& input, - const std::vector& split, - int64_t axis) { - Assert(split.size() > 0, - "OnnxHelper::Split requires the size of parameter split > 0."); - std::vector outputs(split.size()); - for (size_t i = 0; i < split.size(); ++i) { - outputs[i] = MapperHelper::Get()->GenName("helper.split"); - } - return Split(input, outputs, split, axis); -} -std::vector OnnxHelper::DtypeAlignment( - const std::vector& input_info, int32_t* out_dtype) { - Assert(input_info.size() > 0, - "OnnxHelper::DtypeAlignment requires the size of input info > 0."); - std::vector input_dtypes; - input_dtypes.reserve(input_info.size()); - for (auto i = 0; i < input_info.size(); ++i) { - input_dtypes.push_back(input_info[i].dtype); - } - int32_t max_index = -1; - for (auto i : input_dtypes) { - if (i > max_index) { - max_index = i; - } - } - *out_dtype = max_index; - std::vector casted_node; - casted_node.reserve(input_info.size()); - for (auto i = 0; i < input_info.size(); ++i) { - std::string cast_name = - AutoCast(input_info[i].name, input_info[i].dtype, max_index); - casted_node.push_back(cast_name); - } - return casted_node; -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/onnx_helper.h b/paddle2onnx/mapper/onnx_helper.h deleted file mode 100644 index 5958eb38f67..00000000000 --- a/paddle2onnx/mapper/onnx_helper.h +++ /dev/null @@ -1,561 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once - -#include - -#include -#include -#include - -#include "paddle2onnx/mapper/register_mapper.h" -#include "paddle2onnx/parser/parser.h" - -namespace paddle2onnx { - -void AddAttribute(std::shared_ptr node, - const std::string& name, const int64_t& value); -void AddAttribute(std::shared_ptr node, - const std::string& name, const float& value); -void AddAttribute(std::shared_ptr node, - const std::string& name, const std::string& value); -void AddAttribute(std::shared_ptr node, - const std::string& name, const std::vector& values); -void AddAttribute(std::shared_ptr node, - const std::string& name, const std::vector& values); -void AddAttribute(std::shared_ptr node, - const std::string& name, - ONNX_NAMESPACE::TensorProto_DataType dtype); - -ONNX_NAMESPACE::TensorProto_DataType GetOnnxDtype(int32_t paddle_dtype); -std::shared_ptr MakeConstant(const std::string& name, - const Weight& weight); -std::shared_ptr MakeValueInfo( - const TensorInfo& info); - -struct QuantizeInfo { - public: - std::vector scale_; - std::vector zeros_; - std::string zeros_node_; - std::string scale_node_; - int64_t quantize_axis_; - - QuantizeInfo() {} - QuantizeInfo(const std::vector& scale, - const std::vector& zeros, const std::string& scale_node, - const std::string& zeros_node, const int64_t& quantize_axis) { - zeros_node_ = zeros_node; - scale_node_ = scale_node; - quantize_axis_ = quantize_axis; - scale_.resize(scale.size()); - memcpy(scale_.data(), scale.data(), scale.size() * sizeof(float)); - zeros_.resize(zeros.size()); - memcpy(zeros_.data(), zeros.data(), zeros.size() * sizeof(int64_t)); - } -}; - -class OnnxHelper { - public: - std::vector> nodes; - std::vector> value_infos; - int32_t opset_version = 7; - // Use updated_params to store params that were changed during conversion - std::map updated_params; - // Use quantize_info to record quantization-related information, scale and - // zero information corresponding to each tensor - std::map quantize_info; - - void Clear() { nodes.clear(); } - - void SetOpsetVersion(int32_t op_v) { opset_version = op_v; } - - int32_t GetOpsetVersion() { return opset_version; } - - template - bool TryGetTensorValue(const std::string& name, std::vector* value); - - std::shared_ptr MakeValueInfo( - const std::string& name, const int32_t& dtype, - std::vector& shape); - - std::shared_ptr MakeNode( - const std::string& op_type, const std::vector& inputs, - const std::vector& outputs); - // we use this function to generate some temporary node - // we do not need to define the outputs, because the outputs - // is generate by MapperHelper, which will make sure there's no - // name confict problem - // the parameter `num_outputs` will define the number of output names - std::shared_ptr MakeNode( - const std::string& op_type, const std::vector& inputs, - int num_outputs = 1); - - template - std::string ConstOfShape(const std::string& input, const std::string& output, - ONNX_NAMESPACE::TensorProto_DataType dtype, T value); - template - std::string ConstOfShape(const std::string& input, - ONNX_NAMESPACE::TensorProto_DataType dtype, T value); - - std::string AutoCast(const std::string& input, int32_t input_paddle_dtype, - int32_t to_paddle_dtype); - std::string AutoCast(const std::string& input, const std::string& output, - int32_t input_paddle_dtype, int32_t to_paddle_dtype); - - // Helper function for PaddlePaddle's shape tensor list inputs - // will cast all data type to int64 - // will make sure all inputs to be 1-D tensor - // will concat them as output - std::string ConcatIndices(const std::vector& indices); - std::vector DtypeAlignment( - const std::vector& input_info, int32_t* out_dtype); - std::string Clip(const std::string& input, const float& min, const float& max, - const int32_t& in_dtype); - std::string Clip(const std::string& input, const std::string& output, - const float& min, const float& max, const int32_t& in_dtype); - std::string Squeeze(const std::string& input, - const std::vector& axes); - std::string Squeeze(const std::string& input, const std::string& output, - const std::vector& axes); - std::string Unsqueeze(const std::string& input, - const std::vector& axes); - std::string Unsqueeze(const std::string& input, const std::string& output, - const std::vector& axes); - std::string Reshape(const std::string& input, const std::string& output, - const std::vector& shape); - std::string Reshape(const std::string& input, - const std::vector& shape); - std::string Flatten(const std::string& input, const std::string& output); - std::string Flatten(const std::string& input); - std::string Slice(const std::string& input, const std::string& output, - const std::vector& axes, - const std::vector& starts, - const std::vector& ends); - std::string Slice(const std::string& input, const std::vector& axes, - const std::vector& starts, - const std::vector& ends); - std::string Concat(const std::vector& input, - const std::string& output, int64_t axis); - std::string Concat(const std::vector& input, int64_t axis); - - std::string Transpose(const std::string& input, const std::string& output, - const std::vector& perm); - std::string Transpose(const std::string& input, - const std::vector& perm); - - std::vector Split(const std::string& input, - const std::vector& outputs, - const std::vector& split, - int64_t axis); - std::vector Split(const std::string& input, - const std::vector& split, - int64_t axis); - - template - std::string Constant(const std::string& output, - ONNX_NAMESPACE::TensorProto_DataType dtype, - const std::vector& value); - template - std::string Constant(ONNX_NAMESPACE::TensorProto_DataType dtype, - const std::vector& value); - template - std::string Constant(const std::string& output, - const std::vector& shape, - ONNX_NAMESPACE::TensorProto_DataType dtype, T value); - template - std::string Constant(const std::vector& shape, - ONNX_NAMESPACE::TensorProto_DataType dtype, T value); - - template - std::string Constant(const std::vector& shape, - ONNX_NAMESPACE::TensorProto_DataType dtype, - std::vector& value); - - template - std::string Assign(const std::string& output, - const ONNX_NAMESPACE::TensorProto_DataType& dtype, - const std::vector& shape, - const std::vector& value); - template - std::string Assign(const ONNX_NAMESPACE::TensorProto_DataType& dtype, - const std::vector& shape, - const std::vector& value); -}; - -template -std::string OnnxHelper::Constant(const std::vector& shape, - ONNX_NAMESPACE::TensorProto_DataType dtype, - std::vector& value) { - auto node = std::make_shared(); - node->set_op_type("Constant"); - auto name = MapperHelper::Get()->GenName("const"); - node->add_output(name); - auto attr = node->add_attribute(); - attr->set_name("value"); - attr->set_type(ONNX_NAMESPACE::AttributeProto::TENSOR); - auto tensor = attr->mutable_t(); - tensor->set_name(name); - - int numel = 1; - for (size_t i = 0; i < shape.size(); ++i) { - tensor->add_dims(shape[i]); - numel *= shape[i]; - } - Assert(numel == value.size(), - "numel and val number is not equal in Constant " - "function."); - tensor->set_data_type(dtype); - if (dtype == ONNX_NAMESPACE::TensorProto::FLOAT) { - std::vector data; - data.reserve(numel); - for (auto& i : value) { - data.push_back(static_cast(i)); - } - tensor->set_raw_data(std::string((const char*)(data.data()), numel * 4)); - } else if (dtype == ONNX_NAMESPACE::TensorProto::DOUBLE) { - std::vector data; - data.reserve(numel); - for (auto& i : value) { - data.push_back(static_cast(i)); - } - tensor->set_raw_data(std::string((const char*)(data.data()), numel * 8)); - } else if (dtype == ONNX_NAMESPACE::TensorProto::INT64) { - std::vector data; - data.reserve(numel); - for (auto& i : value) { - data.push_back(static_cast(i)); - } - tensor->set_raw_data(std::string((const char*)(data.data()), numel * 8)); - } else if (dtype == ONNX_NAMESPACE::TensorProto::BOOL) { - bool* data = new bool[numel]; - for (size_t i = 0; i < numel; ++i) { - data[i] = static_cast(value[i]); - } - tensor->set_raw_data(std::string((const char*)(data), numel)); - delete[] data; - } else { - Assert(false, - "Only support data type of BOOL/FLOAT/DOUBLE/INT64 in Constant " - "function."); - } - nodes.push_back(node); - return node->output(0); -} - -template -std::string OnnxHelper::Constant(const std::string& output, - ONNX_NAMESPACE::TensorProto_DataType dtype, - const std::vector& value) { - auto node = std::make_shared(); - node->set_op_type("Constant"); - node->add_output(output); - auto attr = node->add_attribute(); - attr->set_name("value"); - attr->set_type(ONNX_NAMESPACE::AttributeProto::TENSOR); - auto tensor = attr->mutable_t(); - tensor->set_name(output); - - int numel = value.size(); - tensor->add_dims(numel); - tensor->set_data_type(dtype); - if (value.size() == 0) { - nodes.push_back(node); - return output; - } - if (dtype == ONNX_NAMESPACE::TensorProto::FLOAT) { - std::vector data; - for (auto& item : value) { - data.push_back(static_cast(item)); - } - tensor->set_raw_data(std::string((const char*)(data.data()), numel * 4)); - } else if (dtype == ONNX_NAMESPACE::TensorProto::DOUBLE) { - std::vector data; - for (auto& item : value) { - data.push_back(static_cast(item)); - } - tensor->set_raw_data(std::string((const char*)(data.data()), numel * 8)); - } else if (dtype == ONNX_NAMESPACE::TensorProto::INT64) { - std::vector data; - for (auto& item : value) { - data.push_back(static_cast(item)); - } - tensor->set_raw_data(std::string((const char*)(data.data()), numel * 8)); - } else if (dtype == ONNX_NAMESPACE::TensorProto::INT32) { - std::vector data; - for (auto& item : value) { - data.push_back(static_cast(item)); - } - tensor->set_raw_data(std::string((const char*)(data.data()), numel * 4)); - } else if (dtype == ONNX_NAMESPACE::TensorProto::BOOL) { - bool* data = new bool[numel]; - for (size_t i = 0; i < numel; ++i) { - data[i] = static_cast(value[i]); - } - tensor->set_raw_data(std::string((const char*)(data), numel)); - delete[] data; - } else if (dtype == ONNX_NAMESPACE::TensorProto::INT8) { - std::vector data; - data.reserve(numel); - for (auto& i : value) { - data.push_back(static_cast(i)); - } - tensor->set_raw_data(std::string((const char*)(data.data()), numel)); - } else { - Assert(false, - "Only support data type of BOOL/FLOAT/DOUBLE/INT32/INT64/INT8 in " - "Constant " - "function."); - } - nodes.push_back(node); - return output; -} - -template -std::string OnnxHelper::Constant(ONNX_NAMESPACE::TensorProto_DataType dtype, - const std::vector& value) { - auto output = MapperHelper::Get()->GenName("helper.constant"); - return Constant(output, dtype, value); -} - -template -std::string OnnxHelper::Constant(const std::string& output, - const std::vector& shape, - ONNX_NAMESPACE::TensorProto_DataType dtype, - T value) { - auto node = std::make_shared(); - node->set_op_type("Constant"); - node->add_output(output); - auto attr = node->add_attribute(); - attr->set_name("value"); - attr->set_type(ONNX_NAMESPACE::AttributeProto::TENSOR); - auto tensor = attr->mutable_t(); - tensor->set_name(output); - - int numel = 1; - for (size_t i = 0; i < shape.size(); ++i) { - tensor->add_dims(shape[i]); - numel *= shape[i]; - } - tensor->set_data_type(dtype); - if (dtype == ONNX_NAMESPACE::TensorProto::FLOAT) { - std::vector data(numel, static_cast(value)); - tensor->set_raw_data(std::string((const char*)(data.data()), numel * 4)); - } else if (dtype == ONNX_NAMESPACE::TensorProto::DOUBLE) { - std::vector data(numel, static_cast(value)); - tensor->set_raw_data(std::string((const char*)(data.data()), numel * 8)); - } else if (dtype == ONNX_NAMESPACE::TensorProto::INT64) { - std::vector data(numel, static_cast(value)); - tensor->set_raw_data(std::string((const char*)(data.data()), numel * 8)); - } else if (dtype == ONNX_NAMESPACE::TensorProto::INT32) { - std::vector data(numel, static_cast(value)); - tensor->set_raw_data(std::string((const char*)(data.data()), numel * 4)); - } else if (dtype == ONNX_NAMESPACE::TensorProto::INT8) { - std::vector data(numel, static_cast(value)); - tensor->set_raw_data(std::string((const char*)(data.data()), numel)); - } else if (dtype == ONNX_NAMESPACE::TensorProto::BOOL) { - bool* data = new bool[numel]; - for (size_t i = 0; i < numel; ++i) { - data[i] = static_cast(value); - } - tensor->set_raw_data(std::string((const char*)(data), numel)); - delete[] data; - } else { - Assert( - false, - "Only support data type of BOOL/FLOAT/DOUBLE/INT32/INT64 in Constant " - "function."); - } - nodes.push_back(node); - return output; -} - -template -std::string OnnxHelper::Constant(const std::vector& shape, - ONNX_NAMESPACE::TensorProto_DataType dtype, - T value) { - auto output = MapperHelper::Get()->GenName("helper.constant"); - return Constant(output, shape, dtype, value); -} - -template -std::string OnnxHelper::ConstOfShape(const std::string& input, - ONNX_NAMESPACE::TensorProto_DataType dtype, - T value) { - auto output = MapperHelper::Get()->GenName("helper.constofshape"); - return ConstOfShape(input, output, dtype, value); -} - -template -std::string OnnxHelper::ConstOfShape(const std::string& input, - const std::string& output, - ONNX_NAMESPACE::TensorProto_DataType dtype, - T value) { - auto node = MakeNode("ConstantOfShape", {input}, {output}); - auto attr = node->add_attribute(); - attr->set_name("value"); - attr->set_type(ONNX_NAMESPACE::AttributeProto::TENSOR); - auto tensor = attr->mutable_t(); - tensor->set_name("tensor_value"); - std::vector shape = {1}; - int numel = 1; - for (size_t i = 0; i < shape.size(); ++i) { - tensor->add_dims(shape[i]); - numel *= shape[i]; - } - tensor->set_data_type(dtype); - if (dtype == ONNX_NAMESPACE::TensorProto::FLOAT) { - std::vector data(numel, static_cast(value)); - tensor->set_raw_data(std::string((const char*)(data.data()), numel * 4)); - } else if (dtype == ONNX_NAMESPACE::TensorProto::DOUBLE) { - std::vector data(numel, static_cast(value)); - tensor->set_raw_data(std::string((const char*)(data.data()), numel * 8)); - } else if (dtype == ONNX_NAMESPACE::TensorProto::INT64) { - std::vector data(numel, static_cast(value)); - tensor->set_raw_data(std::string((const char*)(data.data()), numel * 8)); - } else if (dtype == ONNX_NAMESPACE::TensorProto::INT32) { - std::vector data(numel, static_cast(value)); - tensor->set_raw_data(std::string((const char*)(data.data()), numel * 4)); - } else { - Assert(false, - "Only support data type of FLOAT/DOUBLE/INT64/INT32 in ConstOfShape " - "function."); - } - return output; -} - -template -std::string OnnxHelper::Assign( - const std::string& output, - const ONNX_NAMESPACE::TensorProto_DataType& dtype, - const std::vector& shape, const std::vector& value) { - auto node = std::make_shared(); - node->set_op_type("Constant"); - node->add_output(output); - auto attr = node->add_attribute(); - attr->set_name("value"); - attr->set_type(ONNX_NAMESPACE::AttributeProto::TENSOR); - auto tensor = attr->mutable_t(); - tensor->set_name(output); - - int numel = std::accumulate(std::begin(shape), std::end(shape), 1, - std::multiplies()); - Assert(numel == value.size(), - "Numel of value not satisfy the input shape while creating contant " - "tensor."); - for (size_t i = 0; i < shape.size(); ++i) { - tensor->add_dims(shape[i]); - } - tensor->set_data_type(dtype); - if (dtype == ONNX_NAMESPACE::TensorProto::FLOAT) { - std::vector data; - for (auto& item : value) { - data.push_back(static_cast(item)); - } - tensor->set_raw_data(std::string((const char*)(data.data()), numel * 4)); - } else if (dtype == ONNX_NAMESPACE::TensorProto::DOUBLE) { - std::vector data; - for (auto& item : value) { - data.push_back(static_cast(item)); - } - tensor->set_raw_data(std::string((const char*)(data.data()), numel * 8)); - } else if (dtype == ONNX_NAMESPACE::TensorProto::INT64) { - std::vector data; - for (auto& item : value) { - data.push_back(static_cast(item)); - } - tensor->set_raw_data(std::string((const char*)(data.data()), numel * 8)); - } else if (dtype == ONNX_NAMESPACE::TensorProto::INT32) { - std::vector data; - for (auto& item : value) { - data.push_back(static_cast(item)); - } - tensor->set_raw_data(std::string((const char*)(data.data()), numel * 4)); - } else { - Assert(false, - "Only support data type of FLOAT/DOUBLE/INT32/INT64 in Constant " - "function."); - } - nodes.push_back(node); - return output; -} - -template -std::string OnnxHelper::Assign( - const ONNX_NAMESPACE::TensorProto_DataType& dtype, - const std::vector& shape, const std::vector& value) { - auto output = MapperHelper::Get()->GenName("helper.constant"); - return Assign(output, dtype, shape, value); -} - -template -bool OnnxHelper::TryGetTensorValue(const std::string& name, - std::vector* value) { - for (auto iter = nodes.begin(); iter != nodes.end(); iter++) { - auto node = *iter; - if (node->op_type() != "Constant") { - continue; - } - if (node->output(0) == name) { - for (auto i = 0; i < node->attribute_size(); i++) { - auto attr = node->attribute(i); - if (attr.name() == "value") { - auto tensor = attr.mutable_t(); - auto dtype = tensor->data_type(); - std::vector shape; - for (int64_t i = 0; i < tensor->dims_size(); i++) { - shape.push_back(tensor->dims(i)); - } - int64_t nums = 1; - for (auto& i : shape) nums *= i; - value->resize(nums); - if (dtype == ONNX_NAMESPACE::TensorProto::INT64) { - std::vector val(nums, 0); - memcpy(val.data(), tensor->raw_data().data(), - nums * sizeof(int64_t)); - value->assign(val.begin(), val.end()); - return true; - } else if (dtype == ONNX_NAMESPACE::TensorProto::INT32) { - std::vector val(nums, 0); - memcpy(val.data(), tensor->raw_data().data(), - nums * sizeof(int32_t)); - value->assign(val.begin(), val.end()); - return true; - } else if (dtype == ONNX_NAMESPACE::TensorProto::FLOAT) { - std::vector val(nums, 0); - memcpy(val.data(), tensor->raw_data().data(), nums * sizeof(float)); - value->assign(val.begin(), val.end()); - return true; - } else if (dtype == ONNX_NAMESPACE::TensorProto::DOUBLE) { - std::vector val(nums, 0); - memcpy(val.data(), tensor->raw_data().data(), - nums * sizeof(double)); - value->assign(val.begin(), val.end()); - return true; - } else { - P2OLogger() << "[WARNING] OnnxHelper function TryGetTensorValue " - "only support get int64_t/int32_t/float/double " - "value from Constant now." - << std::endl; - return false; - } - } - } - } - } - return false; -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/quantize/dequantize_linear.cc b/paddle2onnx/mapper/quantize/dequantize_linear.cc deleted file mode 100644 index 9bf7797fdc8..00000000000 --- a/paddle2onnx/mapper/quantize/dequantize_linear.cc +++ /dev/null @@ -1,183 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/quantize/dequantize_linear.h" - -namespace paddle2onnx { -REGISTER_MAPPER(dequantize_linear, DequantizeLinearMapper) - -int32_t DequantizeLinearMapper::GetMinOpset(bool verbose) { - if (!IsConstantInput("Scale")) { - Error() << "Input `Scale` requires to be a constant tensor." << std::endl; - return -1; - } - std::vector scales; - if (!TryGetInputValue("Scale", &scales)) { - Error() << "Failed to read tensor value of `Scale`." << std::endl; - return -1; - } - if (bit_length_ != 8) { - Error() << "Only support bit_length = 8." << std::endl; - return -1; - } - if (scales.size() > 1) { - auto x_info = GetInput("X"); - if (x_info[0].shape[quant_axis_] != scales.size()) { - Error() << "Scale size must equal to the size of input quantize axis." - << std::endl; - return -1; - } - Logger(verbose, 13) << "While size of scales greater than 1, " - << RequireOpset(13) << std::endl; - return 13; - } - auto x_info = GetInput("X"); - auto x_shape = x_info[0].shape; - if (x_shape.size() == 2) { - if (quant_axis_ != 1) { - Error() << "When the rank of input is 2, the attribute quant_axis " - "requires to be 1." - << std::endl; - return -1; - } - } else if (x_shape.size() == 4) { - if (!(quant_axis_ == 1 || quant_axis_ == 0)) { - Error() << "When the rank of input is 4, the attribute quant_axis " - "requires to be 0 or 1." - << std::endl; - return -1; - } - } - - Logger(verbose, 10) << RequireOpset(10) << std::endl; - return 10; -} - -void DequantizeLinearMapper::ConvertInt8ToFp32( - const std::vector &onnx_scales, std::vector *weight) { - auto x_info = GetInput("X"); - auto x_shape = x_info[0].shape; - if (x_shape.size() == 2) { - for (auto j = 0; j < x_shape[1]; ++j) { - float scale_value = 0; - if (onnx_scales.size() == 1) { - scale_value = onnx_scales[0]; - } else { - scale_value = onnx_scales[j]; - } - for (auto i = 0; i < x_shape[0]; ++i) { - auto offset = i * x_shape[1] + j; - (*weight)[offset] *= scale_value; - } - } - } else if (x_shape.size() == 4) { - if (quant_axis_ == 0) { - auto inner_offset = 1; - for (auto i : x_shape) { - inner_offset *= i; - } - inner_offset /= x_shape[0]; - for (int i = 0; i < x_shape[0]; ++i) { - float scale_value = 0; - if (onnx_scales.size() == 1) { - scale_value = onnx_scales[0]; - } else { - scale_value = onnx_scales[i]; - } - for (auto j = 0; j < inner_offset; ++j) { - auto offset = i * inner_offset + j; - (*weight)[offset] *= scale_value; - } - } - } else { - auto inner_offset = x_shape[2] * x_shape[3]; - auto outter_offset = x_shape[1] * inner_offset; - for (auto i = 0; i < x_shape[0]; ++i) { - for (auto j = 0; j < x_shape[1]; ++j) { - float scale_value = 0; - if (onnx_scales.size() == 1) { - scale_value = onnx_scales[0]; - } else { - scale_value = onnx_scales[j]; - } - for (auto k = 0; k < inner_offset; k++) { - auto offset = i * outter_offset + j * inner_offset + k; - (*weight)[offset] *= scale_value; - } - } - } - } - } -} - -void DequantizeLinearMapper::Opset10() { - auto x_info = GetInput("X"); - auto x_shape = x_info[0].shape; - std::vector scales; - Assert(TryGetInputValue("Scale", &scales), - "Failed to read tensor value of `Scale`."); - std::vector onnx_scales; - onnx_scales.reserve(scales.size()); - for (auto &i : scales) { - onnx_scales.push_back(i / 127); - } - std::vector onnx_zeros(onnx_scales.size(), 0); - std::string scale_node, zero_node; - if (onnx_zeros.size() == 1) { - scale_node = helper_->Constant({}, ONNX_NAMESPACE::TensorProto::FLOAT, - onnx_scales[0]); - zero_node = - helper_->Constant({}, ONNX_NAMESPACE::TensorProto::INT8, onnx_zeros[0]); - } else { - scale_node = - helper_->Constant(ONNX_NAMESPACE::TensorProto::FLOAT, onnx_scales); - zero_node = - helper_->Constant(ONNX_NAMESPACE::TensorProto::INT8, onnx_zeros); - } - - std::vector weight; - TryGetInputValue("X", &weight); - if (weight.empty()) { - auto node = helper_->MakeNode("DequantizeLinear", - {x_info[0].name, scale_node, zero_node}, - {GetOutput("Y")[0].name}); - if (helper_->GetOpsetVersion() >= 13) { - AddAttribute(node, "axis", quant_axis_); - } - QuantizeInfo quantize_info(onnx_scales, onnx_zeros, scale_node, zero_node, - quant_axis_); - helper_->quantize_info[GetOutput("Y")[0].name] = quantize_info; - return; - } - ConvertInt8ToFp32(onnx_scales, &weight); - - QuantizeInfo quantize_info(onnx_scales, onnx_zeros, scale_node, zero_node, - quant_axis_); - helper_->quantize_info[x_info[0].name] = quantize_info; - Weight fp32_weight; - fp32_weight.set(P2ODataType::FP32, x_shape, weight); - helper_->updated_params[x_info[0].name] = fp32_weight; - auto node = helper_->MakeNode("QuantizeLinear", - {x_info[0].name, scale_node, zero_node}); - if (helper_->GetOpsetVersion() >= 13) { - AddAttribute(node, "axis", quant_axis_); - } - auto dq_node = helper_->MakeNode("DequantizeLinear", - {node->output(0), scale_node, zero_node}, - {GetOutput("Y")[0].name}); - if (helper_->GetOpsetVersion() >= 13) { - AddAttribute(dq_node, "axis", quant_axis_); - } -} -} // namespace paddle2onnx \ No newline at end of file diff --git a/paddle2onnx/mapper/quantize/dequantize_linear.h b/paddle2onnx/mapper/quantize/dequantize_linear.h deleted file mode 100755 index e49ed1d2245..00000000000 --- a/paddle2onnx/mapper/quantize/dequantize_linear.h +++ /dev/null @@ -1,42 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class DequantizeLinearMapper : public Mapper { - public: - DequantizeLinearMapper(const PaddleParser& p, OnnxHelper* helper, - int64_t block_id, int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("quant_axis", &quant_axis_); - GetAttr("bit_length", &bit_length_); - if (quant_axis_ == -1) { - quant_axis_ = 1; - } - } - - int32_t GetMinOpset(bool verbose = false); - void Opset10(); - - private: - void ConvertInt8ToFp32(const std::vector& onnx_scales, - std::vector* weight); - int64_t quant_axis_ = 1; - int64_t bit_length_ = 8; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/quantize/quantize_linear.cc b/paddle2onnx/mapper/quantize/quantize_linear.cc deleted file mode 100755 index 6ff1c188061..00000000000 --- a/paddle2onnx/mapper/quantize/quantize_linear.cc +++ /dev/null @@ -1,88 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/quantize/quantize_linear.h" - -namespace paddle2onnx { -REGISTER_MAPPER(quantize_linear, QuantizeLinearMapper) - -int32_t QuantizeLinearMapper::GetMinOpset(bool verbose) { - if (!IsConstantInput("Scale")) { - Error() << "Input `Scale` requires to be a constant tensor." << std::endl; - return -1; - } - std::vector scales; - if (!TryGetInputValue("Scale", &scales)) { - Error() << "Failed to read tensor value of `Scale`." << std::endl; - return -1; - } - if (bit_length_ != 8) { - Error() << "Only support bit_length = 8." << std::endl; - return -1; - } - if (round_type_ != 0) { - Error() << "The round_type attr of quantize_linear must be 0." << std::endl; - return -1; - } - if (scales.size() > 1) { - auto x_info = GetInput("X"); - if (x_info[0].shape[quant_axis_] != scales.size()) { - Error() << "Scale size must equal to the size of input quantize axis." - << std::endl; - return -1; - } - Logger(verbose, 13) << "While size of scales greater than 1, " - << RequireOpset(13) << std::endl; - return 13; - } - Logger(verbose, 10) << RequireOpset(10) << std::endl; - return 10; -} - -void QuantizeLinearMapper::Opset10() { - auto x_info = GetInput("X"); - std::vector scales; - Assert(TryGetInputValue("Scale", &scales), - "Failed to read tensor value of `Scale`."); - std::vector onnx_scales; - onnx_scales.reserve(scales.size()); - for (auto i : scales) { - onnx_scales.push_back(i / 127); - } - std::vector onnx_zeros(onnx_scales.size(), 0); - - std::string scale_node, zero_node; - if (onnx_scales.size() == 1) { - scale_node = helper_->Constant({}, ONNX_NAMESPACE::TensorProto::FLOAT, - onnx_scales[0]); - zero_node = - helper_->Constant({}, ONNX_NAMESPACE::TensorProto::INT8, onnx_zeros[0]); - } else { - scale_node = - helper_->Constant(ONNX_NAMESPACE::TensorProto::FLOAT, onnx_scales); - zero_node = - helper_->Constant(ONNX_NAMESPACE::TensorProto::INT8, onnx_zeros); - } - - auto node = helper_->MakeNode("QuantizeLinear", - {x_info[0].name, scale_node, zero_node}, - {GetOutput("Y")[0].name}); - if (helper_->GetOpsetVersion() >= 13) { - AddAttribute(node, "axis", quant_axis_); - } - QuantizeInfo quantize_info(onnx_scales, onnx_zeros, scale_node, zero_node, - quant_axis_); - helper_->quantize_info[x_info[0].name] = quantize_info; -} -} // namespace paddle2onnx \ No newline at end of file diff --git a/paddle2onnx/mapper/quantize/quantize_linear.h b/paddle2onnx/mapper/quantize/quantize_linear.h deleted file mode 100644 index a209fdbb025..00000000000 --- a/paddle2onnx/mapper/quantize/quantize_linear.h +++ /dev/null @@ -1,45 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class QuantizeLinearMapper : public Mapper { - public: - QuantizeLinearMapper(const PaddleParser& p, OnnxHelper* helper, - int64_t block_id, int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("quant_axis", &quant_axis_); - GetAttr("bit_length", &bit_length_); - if (quant_axis_ == -1) { - quant_axis_ = 1; - } - if (HasAttr("round_type")) { - GetAttr("round_type", &round_type_); - } - } - - int32_t GetMinOpset(bool verbose = false); - void Opset10(); - - private: - int64_t round_type_ = 0; // 0: rounding to nearest ties to even. 1: rounding - // to nearest ties away from zero. - int64_t quant_axis_ = 1; - int64_t bit_length_ = 8; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/quantize_helper.cc b/paddle2onnx/mapper/quantize_helper.cc deleted file mode 100644 index 84446d28741..00000000000 --- a/paddle2onnx/mapper/quantize_helper.cc +++ /dev/null @@ -1,1231 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/quantize_helper.h" - -namespace paddle2onnx { - -void QuantizeModelProcessor::RemoveNodeByName(const std::string& name, - const bool& update_io) { - if (name.empty()) { - return; - } - for (auto iter = nodes_->begin(); iter != nodes_->end(); iter++) { - if ((*iter)->name() == name) { - std::string input_name = (*iter)->input(0); - std::string output_name = (*iter)->output(0); - nodes_->erase(iter); - if (update_io) { - ReplaceInputOfAllNodes(output_name, input_name); - } - return; - } - } -} - -void QuantizeModelProcessor::ReplaceInputOfAllNodes( - const std::string& old_name, const std::string& new_name, - const std::vector>& - except_nodes) { - auto iter = name2node_dict_.find(old_name); - std::vector> need_rename_nodes; - // after replace all old_name to new_name, replace the quantize_info of new - // name with the quantize_info of old name - auto quantize_info_iter = helper_->quantize_info.find(old_name); - if (quantize_info_iter != helper_->quantize_info.end()) { - helper_->quantize_info[new_name] = helper_->quantize_info[old_name]; - } - - if (iter != name2node_dict_.end()) { - need_rename_nodes = iter->second; - } - for (auto& node : need_rename_nodes) { - auto iter = std::find(except_nodes.begin(), except_nodes.end(), node); - if (iter != except_nodes.end()) { - continue; - } - for (size_t i = 0; i < node->input_size(); ++i) { - if (node->input(i) == old_name) { - node->set_input(i, new_name); - } - } - } -} - -void QuantizeModelProcessor::UpdateInputNameToNodes() { - name2node_dict_.clear(); - for (auto& node : *nodes_) { - for (size_t i = 0; i < node->input_size(); ++i) { - std::string node_input = node->input(i); - if (name2node_dict_.find(node_input) != name2node_dict_.end()) { - name2node_dict_[node_input].push_back(node); - } else { - name2node_dict_[node_input] = {node}; - } - } - } -} - -void QuantizeModelProcessor::ProcessQuantizeModel( - std::vector>* parameters, - std::vector>* inputs, - std::vector>* outputs, - std::vector>* nodes, - OnnxHelper* helper, const std::string& deploy_backend, - const PaddleParser& parser, std::string* calibration_cache) { - // Determine whether the model contains quantization related OPs, if not, exit - // directly - bool quantized_model = false; - for (auto& node : *nodes) { - if (node->op_type() == "QuantizeLinear" || - node->op_type() == "DequantizeLinear") { - quantized_model = true; - break; - } - } - if (!quantized_model) { - return; - } - parser_ = &parser; - helper_ = helper; - parameters_ = parameters; - inputs_ = inputs; - outputs_ = outputs; - nodes_ = nodes; - P2OLogger() << "[Info] Quantize model deploy backend is: " << deploy_backend - << std::endl; - // Determine the format of the exported ONNX quantization model according to - // the deploy_backend - if (deploy_backend == "others") { - // If deploy_backend is others, the quantization model is exported as a - // float model + quantization table. - RemoveAllQuantizeOps(); - std::ofstream outfile; - outfile.open("max_range.txt", std::ios::out); - if (!outfile.is_open()) { - P2OLogger() << "[WARNING] Quantize model processer failed to write range " - "information in current location." - << std::endl; - return; - } - for (auto iter = helper_->quantize_info.begin(); - iter != helper_->quantize_info.end(); iter++) { - std::string log = iter->first; - auto scale = iter->second.scale_; - if (scale.size() == 1) { - log = log + ": " + std::to_string(scale[0] * 127); - outfile << log << std::endl; - } - } - outfile.close(); - } else if (deploy_backend == "onnxruntime") { - // When deploy_backend is ONNXRuntime, use the follow four steps to process: - // 1. broadcast quantize info - // 2. remove all quantize ops - // 3. merge conv and add - // 4. merge conv and bn - // 5. add Q and DQ according ONNXRuntime quantize OP fuse patten. - // 6. use topo sort in nodes - QuantizeInfoBroadcast(); - RemoveAllQuantizeOps(); - MergeConvAdd(); - MergeConvBN(); - AddQDQForORT(); - SortNodes(); - } else if (deploy_backend == "tensorrt") { - // When deploy_backend is TensorRT, use the follow four steps to process: - // For Explicit Quantization - // 1. broadcast quantize info - // 2. remove all quantize ops - // 3. add Q and DQ before conv and matmul. - // 4. use topo sort in nodes - - // For Implicit Quantization - // 1. remove all quantize ops - // 2. broadcast quantize info - // 3. save float onnx model and alibration.cache - QuantizeInfoBroadcast(); - RemoveAllQuantizeOps(); - // Add qdq for Explicit Quantization - // AddTrtQDQ(); - // SortNodes(); - - // Genarate calibration.cache for Implicit Quantization - // convert float to hex - GenerateCache(calibration_cache); - } else if (deploy_backend == "rknn") { - // When deploy_backend is RKNN, use the follow four steps to process: - // 1. broadcast quantize info - // 2. remove all quantize ops - // 3. merge conv and add - // 4. merge conv and bn - // 5. add Q and DQ - // 6. use topo sort in nodes - QuantizeInfoBroadcast(); - RemoveAllQuantizeOps(); - RemoveIdentityOp(); - MergeConvAdd(); - MergeConvBN(); - AddQDQForRKNN(); - SortNodes(); - } else { - Assert(false, - "[QuantizeModelProcessor] Only support 'onnxruntime' / 'tensorrt' " - "/ 'others' as " - "backend now, but now the backend is: " + - deploy_backend + "."); - } -} - -void QuantizeModelProcessor::RemoveIdentityOp() { - UpdateInputNameToNodes(); - auto iter = nodes_->begin(); - while (iter != nodes_->end()) { - auto node = *iter; - if (node->op_type() == "Identity" && !ConnectToOutput(node->output(0))) { - RemoveNodeByName(node->name()); - } else { - iter++; - } - } -} - -void QuantizeModelProcessor::AddQDQForRKNN() { - UpdateInputNameToNodes(); - supported_quantize_type_ = {"Abs", - "Acos", - "Add", - "Asin", - "Atan", - "AveragePool", - "BatchNormalization", - "Ceil", - "Clip", - "Conv", - "ConvTranspose", - "Cos", - "Cosh", - "Concat", - "Elu", - "Erf", - "Exp", - "Floor", - "Gemm", - "HardSigmoid", - "InstanceNormalization", - "IsInf", - "IsNaN", - "Log", - "MaxPool", - "Mul", - "Neg", - "ReduceMean", - "Relu", - "Resize", - "Round", - "Sigmoid", - "Sin", - "Sinh", - "Split", - "Sqrt", - "Tan", - "MatMul", - "Tanh"}; - for (auto iter = nodes_->begin(); iter < nodes_->end(); iter++) { - auto node = *iter; - auto type_iter = std::find(supported_quantize_type_.begin(), - supported_quantize_type_.end(), node->op_type()); - if (!supported_quantize_type_.empty() && - type_iter == supported_quantize_type_.end()) { - continue; - } - // Add Miss scale for Mul and MatMul - if (node->op_type() == "Mul" || node->op_type() == "MatMul" || - node->op_type() == "Add") { - for (size_t i = 0; i < node->input_size(); ++i) { - std::string node_input = node->input(i); - if (helper_->quantize_info.find(node_input) != - helper_->quantize_info.end()) { - continue; - } - std::vector weight; - if (!GetTensorByName(node_input, &weight)) { - continue; - } - std::vector weight_shape; - GetTensorShape(node_input, &weight_shape); - int64_t quantize_axis = 1; - if (node->op_type() == "Add") { - quantize_axis = 0; - } - std::vector scale; - std::vector zeros; - - if (node->op_type() == "Add" || weight_shape.size() == 1 || - weight_shape.empty()) { - GetTensorWiseQuantizeInfo(weight, &scale, &zeros); - } else { - GetChannelWiseQuantizeInfo(weight, weight_shape, quantize_axis, - &scale, &zeros); - } - std::string weight_scale_node, weight_zero_node; - if (scale.size() == 1) { - weight_scale_node = helper_->Constant( - {}, ONNX_NAMESPACE::TensorProto::FLOAT, scale[0]); - weight_zero_node = helper_->Constant( - {}, ONNX_NAMESPACE::TensorProto::INT8, zeros[0]); - } else { - weight_scale_node = - helper_->Constant(ONNX_NAMESPACE::TensorProto::FLOAT, scale); - weight_zero_node = - helper_->Constant(ONNX_NAMESPACE::TensorProto::INT8, zeros); - } - QuantizeInfo weight_quantize_info(scale, zeros, weight_scale_node, - weight_zero_node, quantize_axis); - helper_->quantize_info[node_input] = weight_quantize_info; - } - } - std::vector tensor_names; - for (size_t i = 0; i < node->input_size(); ++i) { - std::string node_input = node->input(i); - tensor_names.push_back(node_input); - } - for (size_t i = 0; i < node->output_size(); ++i) { - std::string node_output = node->output(i); - tensor_names.push_back(node_output); - } - if (!CanBeQuantize(tensor_names)) { - continue; - } - for (auto& name : tensor_names) { - AppendQuantizeTensor(name); - } - } - - // update name2node_dict for the change of Relu op. - UpdateInputNameToNodes(); - // Add QDQ in model - AddQDQInModel(tensors_to_be_quantize); -} - -void QuantizeModelProcessor::GenerateCache(std::string* calibration_cache) { - union { - float f; - unsigned char farray[4]; - } un; - *calibration_cache += "TRT-8XXX-EntropyCalibration2 \n"; - for (auto iter = helper_->quantize_info.rbegin(); - iter != helper_->quantize_info.rend(); iter++) { - std::string tensor_name = iter->first; - QuantizeInfo quantize_info = iter->second; - if (quantize_info.scale_.size() == 1) { - float val = quantize_info.scale_[0]; - un.f = val; - *calibration_cache += (tensor_name + ": "); - std::stringstream enc; - for (int64_t i = 3; i >= 0; i--) { - enc << std::hex << std::setw(2) << std::setfill('0') - << (int)(un.farray[i]); - } - *calibration_cache = *calibration_cache + enc.str() + "\n"; - } - } -} -// In TensorRT, all quantized op: Conv, ConvTranspose, liner(MatMul), MaxPool, -// AvgPool, AdaptiveAvgPool, rnn(not support now) -// https://github.com/NVIDIA/TensorRT/tree/main/tools/pytorch-quantization/pytorch_quantization/nn/modules -void QuantizeModelProcessor::AddTrtQDQ() { - UpdateInputNameToNodes(); - std::vector - quantize_tensors; // save the tensor names that need add quantize ops - std::vector pool_types = {"MaxPool", "AvgPool", - "AdaptiveAvgPool"}; - for (auto iter = nodes_->begin(); iter < nodes_->end(); iter++) { - quantize_tensors.clear(); - auto node = *iter; - if (node->op_type() == "Conv" || node->op_type() == "ConvTranspose") { - std::vector tensor_names = {node->input(0), node->input(1)}; - if (!CanBeQuantize(tensor_names)) { - continue; - } - quantize_tensors = tensor_names; - } - if (node->op_type() == "MatMul") { - std::vector tensor_names = {node->input(0), node->input(1)}; - for (auto& name : tensor_names) { - if (helper_->quantize_info.find(name) != helper_->quantize_info.end()) { - continue; - } - std::vector matmul_weight; - if (!GetTensorByName(name, &matmul_weight)) { - continue; - } - std::vector matmul_weight_shape; - if (!GetTensorShape(name, &matmul_weight_shape)) { - continue; - } - int64_t quantize_axis = 1; - std::vector scale; - std::vector zeros; - GetChannelWiseQuantizeInfo(matmul_weight, matmul_weight_shape, - quantize_axis, &scale, &zeros); - auto scale_node = - helper_->Constant(ONNX_NAMESPACE::TensorProto::FLOAT, scale); - auto zero_node = - helper_->Constant(ONNX_NAMESPACE::TensorProto::INT8, zeros); - QuantizeInfo matmul_weight_quantize_info(scale, zeros, scale_node, - zero_node, quantize_axis); - helper_->quantize_info[name] = matmul_weight_quantize_info; - } - if (!CanBeQuantize(tensor_names)) { - continue; - } - quantize_tensors = tensor_names; - } - auto type_iter = - std::find(pool_types.begin(), pool_types.end(), node->op_type()); - if (type_iter != pool_types.end()) { - std::vector tensor_names = {node->input(0)}; - if (!CanBeQuantize(tensor_names)) { - continue; - } - quantize_tensors = tensor_names; - } - - std::string negative_scale_tensor = ""; - for (std::string& name : quantize_tensors) { - Assert( - helper_->quantize_info.find(name) != helper_->quantize_info.end(), - "[QuantizeModelProcessor] Can not find quantize info for tensor: " + - name); - QuantizeInfo quantize_info = helper_->quantize_info[name]; - std::vector scales = quantize_info.scale_; - for (auto& i : scales) { - if (i <= 1e-10) { - negative_scale_tensor = negative_scale_tensor + " " + name; - } - } - } - if (negative_scale_tensor.size() > 0) { - P2OLogger() - << "[Warning] The scale of tensors: [ " + negative_scale_tensor + - " ] contains negative scale, so this OP will not be quantized." - << std::endl; - continue; - } - // An OP requires a separate quantize op - for (std::string& name : quantize_tensors) { - if (IsGraphOutput(name)) { - continue; - } - QuantizeInfo quantize_info = helper_->quantize_info[name]; - std::string scale_node = quantize_info.scale_node_; - std::string zeros_node = quantize_info.zeros_node_; - int64_t quantize_axis = quantize_info.quantize_axis_; - auto q_node = - helper_->MakeNode("QuantizeLinear", {name, scale_node, zeros_node}); - if (helper_->GetOpsetVersion() >= 13) { - AddAttribute(q_node, "axis", quantize_axis); - } - auto dq_node = helper_->MakeNode( - "DequantizeLinear", {q_node->output(0), scale_node, zeros_node}); - if (helper_->GetOpsetVersion() >= 13) { - AddAttribute(dq_node, "axis", quantize_axis); - } - for (size_t i = 0; i < node->input_size(); ++i) { - if (node->input(i) == name) { - node->set_input(i, dq_node->output(0)); - } - } - } - } -} - -// According to: -// https://github.com/microsoft/onnxruntime/blob/master/onnxruntime/core/optimizer/qdq_transformer/selectors_actions/qdq_selector_action_transformer.cc -void QuantizeModelProcessor::AddQDQForORT() { - UpdateInputNameToNodes(); - supported_quantize_type_ = {"Conv", "MatMul", "Mul", "Sigmoid", - "Add", "LeakyRelu", "Relu"}; - for (auto iter = nodes_->begin(); iter < nodes_->end(); iter++) { - auto node = *iter; - auto type_iter = std::find(supported_quantize_type_.begin(), - supported_quantize_type_.end(), node->op_type()); - if (!supported_quantize_type_.empty() && - type_iter == supported_quantize_type_.end()) { - continue; - } - // Here we only add Relu, Conv, mul and matmul, all tensors should add Q and - // DQ will be saved in tensors_to_be_quantize - if (node->op_type() == "Relu") { - std::vector tensor_names = {node->input(0), node->output(0)}; - if (!CanBeQuantize(tensor_names)) { - continue; - } - node->set_op_type("LeakyRelu"); - AddAttribute(node, "alpha", static_cast(0.0)); - for (auto& name : tensor_names) { - AppendQuantizeTensor(name); - } - } - if (node->op_type() == "LeakyRelu") { - std::vector tensor_names = {node->input(0), node->output(0)}; - if (!CanBeQuantize(tensor_names)) { - continue; - } - for (auto& name : tensor_names) { - AppendQuantizeTensor(name); - } - } - if (node->op_type() == "Add") { - std::vector tensor_names = {node->input(0), node->input(1), - node->output(0)}; - if (!CanBeQuantize(tensor_names)) { - continue; - } - for (auto& name : tensor_names) { - AppendQuantizeTensor(name); - } - } - if (node->op_type() == "Sigmoid") { - std::vector tensor_names = {node->input(0), node->output(0)}; - if (!CanBeQuantize(tensor_names)) { - continue; - } - for (auto& name : tensor_names) { - AppendQuantizeTensor(name); - } - } - if (node->op_type() == "Conv") { - std::vector tensor_names = {node->input(0), node->input(1), - node->output(0)}; - if (node->input_size() == 3) { - tensor_names.push_back(node->input(2)); - } - if (!CanBeQuantize(tensor_names, {2})) { - continue; - } - for (auto& name : tensor_names) { - AppendQuantizeTensor(name); - } - } - if (node->op_type() == "MatMul") { - std::vector tensor_names = {node->input(0), node->input(1), - node->output(0)}; - for (auto& name : tensor_names) { - if (helper_->quantize_info.find(name) != helper_->quantize_info.end()) { - continue; - } - std::vector matmul_weight; - if (!GetTensorByName(name, &matmul_weight)) { - continue; - } - std::vector matmul_weight_shape; - if (!GetTensorShape(name, &matmul_weight_shape)) { - continue; - } - int64_t quantize_axis = 1; - std::vector scale; - std::vector zeros; - GetChannelWiseQuantizeInfo(matmul_weight, matmul_weight_shape, - quantize_axis, &scale, &zeros); - auto scale_node = - helper_->Constant(ONNX_NAMESPACE::TensorProto::FLOAT, scale); - auto zero_node = - helper_->Constant(ONNX_NAMESPACE::TensorProto::INT8, zeros); - QuantizeInfo matmul_weight_quantize_info(scale, zeros, scale_node, - zero_node, quantize_axis); - helper_->quantize_info[name] = matmul_weight_quantize_info; - } - if (!CanBeQuantize(tensor_names)) { - tensor_names.pop_back(); - if (!CanBeQuantize(tensor_names)) { - continue; - } - } - for (auto& name : tensor_names) { - AppendQuantizeTensor(name); - } - } - if (node->op_type() == "Mul") { - std::vector tensor_names = {node->input(0), node->input(1), - node->output(0)}; - if (!CanBeQuantize(tensor_names)) { - continue; - } - for (auto& name : tensor_names) { - AppendQuantizeTensor(name); - } - } - } - // update name2node_dict for the change of Relu op. - UpdateInputNameToNodes(); - // Add QDQ in model - AddQDQInModel(tensors_to_be_quantize); -} - -void QuantizeModelProcessor::AddQDQInModel( - const std::vector& tensors_to_be_quantize) { - // add Q and DQ according to tensors_to_be_quantize - for (auto& name : tensors_to_be_quantize) { - if (IsGraphOutput(name)) { - continue; - } - Assert(helper_->quantize_info.find(name) != helper_->quantize_info.end(), - "[QuantizeModelProcessor] Can not find quantize info for tensor: " + - name); - QuantizeInfo quantize_info = helper_->quantize_info[name]; - std::string scale_node = quantize_info.scale_node_; - std::string zeros_node = quantize_info.zeros_node_; - int64_t quantize_axis = quantize_info.quantize_axis_; - auto iter = std::find(only_dequantize_tensors.begin(), - only_dequantize_tensors.end(), name); - if (iter != only_dequantize_tensors.end()) { - // if only add DequantizeLinear - std::vector scale = quantize_info.scale_; - std::vector bias; - Assert(GetTensorByName(name, &bias), - "[QuantizeModelProcessor] Can not find bias value: " + name); - std::vector new_bias(bias.size(), 0); - for (int64_t i = 0; i < bias.size(); i++) { - float scale_val = scale.size() == 1 ? scale[0] : scale[i]; - new_bias[i] = rint(bias[i] / scale_val); - } - Weight updated_bias; - std::vector bias_shape = {static_cast(new_bias.size())}; - updated_bias.set(P2ODataType::INT32, bias_shape, new_bias); - helper_->updated_params[name] = updated_bias; - auto dq_node = - helper_->MakeNode("DequantizeLinear", {name, scale_node, zeros_node}); - if (helper_->GetOpsetVersion() >= 13) { - AddAttribute(dq_node, "axis", quantize_axis); - } - ReplaceInputOfAllNodes(name, dq_node->output(0)); - } else { - // Handle the following situations - // conv conv - // / | \ -> / \ - // conv conv scale DQD scale - // / \ - // conv conv - std::vector> except_nodes; - auto next_nodes = name2node_dict_[name]; - if (next_nodes.size() > 1) { - for (auto& node : next_nodes) { - auto iter = - std::find(supported_quantize_type_.begin(), - supported_quantize_type_.end(), node->op_type()); - if (iter == supported_quantize_type_.end()) { - except_nodes.push_back(node); - } - } - } - // When all the outputs of this tensor cannot be renamed, - // it means that the quantization OP will be merged - if (next_nodes.size() == except_nodes.size()) { - except_nodes.clear(); - } - auto q_node = - helper_->MakeNode("QuantizeLinear", {name, scale_node, zeros_node}); - if (helper_->GetOpsetVersion() >= 13) { - AddAttribute(q_node, "axis", quantize_axis); - } - auto dq_node = helper_->MakeNode( - "DequantizeLinear", {q_node->output(0), scale_node, zeros_node}); - if (helper_->GetOpsetVersion() >= 13) { - AddAttribute(dq_node, "axis", quantize_axis); - } - ReplaceInputOfAllNodes(name, dq_node->output(0), except_nodes); - } - } -} - -void QuantizeModelProcessor::MergeConvBN() { - UpdateInputNameToNodes(); - for (auto iter = nodes_->begin(); iter < nodes_->end(); iter++) { - auto conv_node = *iter; - if (conv_node->op_type() != "Conv") { - continue; - } - - bool act_has_quantize_info = - helper_->quantize_info.find(conv_node->input(0)) != - helper_->quantize_info.end(); - if (!act_has_quantize_info) { - continue; - } - auto next_nodes = name2node_dict_[conv_node->output(0)]; - - if (next_nodes.size() > 1 || IsGraphOutput(conv_node->output(0))) { - continue; - } - - auto bn_node = next_nodes[0]; - if (bn_node->op_type() != "BatchNormalization" || - IsGraphOutput(bn_node->output(0))) { - continue; - } - - std::vector conv_weight; - Assert(GetTensorByName(conv_node->input(1), &conv_weight), - "Can not get " + conv_node->input(1) + " from Conv."); - - std::vector bn_scale; - Assert(GetTensorByName(bn_node->input(1), &bn_scale), - "Can not get " + bn_node->input(1) + " from BN."); - - std::vector bn_bias; - Assert(GetTensorByName(bn_node->input(2), &bn_bias), - "Can not get " + bn_node->input(2) + " from BN."); - - std::vector bn_mean; - Assert(GetTensorByName(bn_node->input(3), &bn_mean), - "Can not get " + bn_node->input(3) + " from BN."); - - std::vector bn_var; - Assert(GetTensorByName(bn_node->input(4), &bn_var), - "Can not get " + bn_node->input(4) + " from BN."); - - float epsilon = 1; - for (auto i = 0; i < bn_node->attribute_size(); i++) { - auto attr = bn_node->attribute(i); - if (attr.name() == "epsilon") { - epsilon = attr.f(); - } - } - - std::vector conv_bias(bn_bias.size(), 0); - std::string conv_bias_node = conv_node->input(1) + ".merged.bias"; - if (conv_node->input_size() == 3) { - conv_bias_node = conv_node->input(2); - conv_bias.clear(); - Assert(GetTensorByName(conv_bias_node, &conv_bias), - "Can not get " + conv_node->input(2) + " in Conv."); - } - - // merge conv and bn - std::vector alpha(bn_scale.size()); - for (int64_t i = 0; i < bn_scale.size(); i++) { - alpha[i] = bn_scale[i] / sqrt(bn_var[i] + epsilon); - } - - std::vector new_bias(bn_scale.size()); - for (int64_t i = 0; i < bn_scale.size(); i++) { - new_bias[i] = - conv_bias[i] * alpha[i] + (bn_bias[i] - bn_mean[i] * alpha[i]); - } - std::vector new_weight(conv_weight.size()); - int64_t offset = conv_weight.size() / bn_bias.size(); - for (int64_t i = 0; i < bn_scale.size(); i++) { - int64_t outter_offset = i * offset; - for (int64_t j = 0; j < offset; j++) { - int64_t index = outter_offset + j; - new_weight[index] = conv_weight[index] * alpha[i]; - } - } - // update weight - std::vector weight_shape; - Assert(GetTensorShape(conv_node->input(1), &weight_shape), - "Can not get the shape of " + conv_node->input(1) + " in Conv."); - Weight updated_conv_weight; - updated_conv_weight.set(P2ODataType::FP32, weight_shape, new_weight); - helper_->updated_params[conv_node->input(1)] = updated_conv_weight; - // update bias - Weight updated_bias_weight; - std::vector bias_shape = {static_cast(new_bias.size())}; - updated_bias_weight.set(P2ODataType::FP32, bias_shape, new_bias); - helper_->updated_params[conv_bias_node] = updated_bias_weight; - AppendQuantizeTensor(conv_bias_node, true); - // update weight scale - auto quantize_info = helper_->quantize_info[conv_node->input(1)]; - std::string scale_node = quantize_info.scale_node_; - std::string zero_node = quantize_info.zeros_node_; - int64_t quantize_axis = quantize_info.quantize_axis_; - RemoveNodeByName(scale_node); - RemoveNodeByName(zero_node); - std::vector scale = quantize_info.scale_; - std::vector new_scale; - std::vector new_zeros; - if (scale.size() == 1) { - GetTensorWiseQuantizeInfo(new_weight, &new_scale, &new_zeros); - } else { - GetChannelWiseQuantizeInfo(new_weight, weight_shape, quantize_axis, - &new_scale, &new_zeros); - } - auto weight_scale_node = - helper_->Constant(ONNX_NAMESPACE::TensorProto::FLOAT, new_scale); - auto weight_zero_node = - helper_->Constant(ONNX_NAMESPACE::TensorProto::INT8, new_zeros); - QuantizeInfo updated_weight_quantize_info(new_scale, new_zeros, - weight_scale_node, - weight_zero_node, quantize_axis); - helper_->quantize_info[conv_node->input(1)] = updated_weight_quantize_info; - // add bias scale and update bias - auto act_quantize_info = helper_->quantize_info[conv_node->input(0)]; - std::vector act_scale = act_quantize_info.scale_; - std::vector bias_scale; - for (int64_t i = 0; i < new_scale.size(); i++) { - bias_scale.push_back(act_scale[0] * new_scale[i]); - } - std::vector bias_zeros(bias_scale.size(), 0); - auto bias_scale_node = - helper_->Constant(ONNX_NAMESPACE::TensorProto::FLOAT, bias_scale); - auto bias_zero_node = - helper_->Constant(ONNX_NAMESPACE::TensorProto::INT32, bias_zeros); - QuantizeInfo bias_quantize_info(bias_scale, bias_zeros, bias_scale_node, - bias_zero_node, 0); - helper_->quantize_info[conv_bias_node] = bias_quantize_info; - if (conv_node->input_size() == 2) { - conv_node->add_input(conv_bias_node); - } - // remove BN op - RemoveNodeByName(bn_node->name()); - } -} - -void QuantizeModelProcessor::MergeConvAdd() { - UpdateInputNameToNodes(); - for (auto iter = nodes_->begin(); iter < nodes_->end(); iter++) { - auto node = *iter; - if (node->op_type() != "Conv") { - continue; - } - // if act input of conv does not have quantize info, continue - bool act_has_quantize_info = helper_->quantize_info.find(node->input(0)) != - helper_->quantize_info.end(); - if (!act_has_quantize_info) { - continue; - } - - // if weight of conv does not have quantize info, continue - bool weight_has_quantize_info = - helper_->quantize_info.find(node->input(1)) != - helper_->quantize_info.end(); - if (!weight_has_quantize_info) { - continue; - } - auto next_nodes = name2node_dict_[node->output(0)]; - - if (next_nodes.size() > 1 || IsGraphOutput(node->output(0))) { - continue; - } - - auto next_node = next_nodes[0]; - if (next_node->op_type() != "Add" || IsGraphOutput(next_node->output(0))) { - continue; - } - std::string reshape_node = node->output(0) == next_node->input(0) - ? next_node->input(1) - : next_node->input(0); - std::vector> before_nodes; - for (auto& node : *nodes_) { - for (size_t i = 0; i < node->output_size(); ++i) { - std::string node_output = node->output(i); - if (node_output == reshape_node) { - before_nodes.push_back(node); - break; - } - } - } - - if (before_nodes.size() != 1 || before_nodes[0]->op_type() != "Reshape") { - continue; - } - - std::string bias_node = before_nodes[0]->input(0); - // continue if bias is not a constant - std::vector bias_val; - if (!GetTensorByName(bias_node, &bias_val)) { - continue; - } - - // continue if shape tensor of reshape op is not a constant - std::vector shape_val; - if (!GetTensorByName(before_nodes[0]->input(1), &shape_val)) { - continue; - } - // continue if shape_val != [1, bias_val.size(), 1, 1] - std::vector target = {1, static_cast(bias_val.size()), 1, - 1}; - if (target != shape_val) { - continue; - } - // remove Reshape op - RemoveNodeByName(before_nodes[0]->name()); - // add scale for bias - std::vector weight_scale = - helper_->quantize_info[node->input(1)].scale_; - std::vector act_scale = - helper_->quantize_info[node->input(0)].scale_; - std::vector bias_scale; - for (int64_t i = 0; i < weight_scale.size(); i++) { - bias_scale.push_back(weight_scale[i] * act_scale[0]); - } - std::vector onnx_zeros(bias_scale.size(), 0); - auto scale_node = - helper_->Constant(ONNX_NAMESPACE::TensorProto::FLOAT, bias_scale); - auto zero_node = - helper_->Constant(ONNX_NAMESPACE::TensorProto::INT32, onnx_zeros); - - QuantizeInfo quantize_info(bias_scale, onnx_zeros, scale_node, zero_node, - 0); - - helper_->quantize_info[bias_node] = quantize_info; - AppendQuantizeTensor(bias_node, true); - node->add_input(bias_node); - RemoveNodeByName(next_node->name()); - } -} - -void QuantizeModelProcessor::SortNodes() { - // return the topo sort of nodes; - // 1. Get i2o_mapper and constant_nodes, i2o_mapper means the node map to its - // all output nodes, constant_nodes save all constant nodes. - // 2. Nodes without output nodes are first saved to new_nodes, and then - // cyclically delete the records of the node in i2o_mapper items, and nodes - // whose output nodes are empty are also saved to new_nodes in turn. - // 3. Store constant nodes in new_nodes. - // 4. Reverse new_nodes, then assign to nodes. - std::map> i2o_mapper; - std::vector> constant_nodes; - std::map> - name2node_mapper; - for (int64_t i = 0; i < nodes_->size(); i++) { - auto node = (*nodes_)[i]; - if (node->op_type() == "Constant") { - constant_nodes.push_back(node); - continue; - } - name2node_mapper[node->name()] = node; - for (int64_t in_index = 0; in_index < node->input_size(); in_index++) { - std::string input = node->input(in_index); - for (int64_t j = 0; j < nodes_->size(); j++) { - if (i == j) { - continue; - } - auto input_node = (*nodes_)[j]; - if (input_node->op_type() == "Constant") { - continue; - } - for (int64_t out_index = 0; out_index < input_node->output_size(); - out_index++) { - if (input == input_node->output(out_index)) { - if (i2o_mapper.find(input_node->name()) == i2o_mapper.end()) { - i2o_mapper[input_node->name()] = {node->name()}; - } else { - auto iter = - std::find(i2o_mapper[input_node->name()].begin(), - i2o_mapper[input_node->name()].end(), node->name()); - if (iter == i2o_mapper[input_node->name()].end()) { - i2o_mapper[input_node->name()].push_back(node->name()); - } - } - } - } - } - } - } - std::vector> new_nodes; - - for (int64_t i = 0; i < nodes_->size(); i++) { - auto node_name = (*nodes_)[i]->name(); - auto node = (*nodes_)[i]; - if (node->op_type() == "Constant") { - continue; - } - if (i2o_mapper.find(node_name) == i2o_mapper.end()) { - new_nodes.push_back(node); - } - } - int64_t index = 0; - while (index < new_nodes.size()) { - auto current_node = new_nodes[index]; - std::string current_node_name = current_node->name(); - for (auto iter = i2o_mapper.begin(); iter != i2o_mapper.end(); iter++) { - std::string input_node_name = iter->first; - std::vector* output_nodes_name = &iter->second; - if (output_nodes_name->empty()) { - continue; - } - auto in_inter = std::find(output_nodes_name->begin(), - output_nodes_name->end(), current_node_name); - if (in_inter != output_nodes_name->end()) { - output_nodes_name->erase(in_inter); - } - if (output_nodes_name->empty()) { - new_nodes.push_back(name2node_mapper[input_node_name]); - } - } - index++; - } - - for (auto& node : constant_nodes) { - new_nodes.push_back(node); - } - std::reverse(new_nodes.begin(), new_nodes.end()); - Assert(nodes_->size() == new_nodes.size(), - "The number of nodes after topological sorting is not equal to the " - "number before sorting"); - *nodes_ = new_nodes; -} - -void QuantizeModelProcessor::RemoveAllQuantizeOps() { - UpdateInputNameToNodes(); - for (auto iter = nodes_->begin(); iter < nodes_->end(); iter++) { - auto node = *iter; - if (node->op_type() != "QuantizeLinear") { - continue; - } - auto next_node_names = name2node_dict_[node->output(0)]; - - if (next_node_names.empty() || !next_node_names[0]->has_op_type() || - next_node_names[0]->op_type() != "DequantizeLinear") { - continue; - } - std::string input_name = node->input(0); - RemoveNodeByName(node->name(), false); - std::string output_name = next_node_names[0]->output(0); - RemoveNodeByName(next_node_names[0]->name(), false); - if (ConnectToOutput(output_name)) { - for (auto pre_iter = nodes_->begin(); pre_iter < nodes_->end(); - pre_iter++) { - auto pre_node = *pre_iter; - for (size_t o_idex = 0; o_idex < pre_node->output_size(); ++o_idex) { - if (pre_node->output(o_idex) == input_name) { - pre_node->set_output(o_idex, output_name); - } - } - } - } else { - ReplaceInputOfAllNodes(output_name, input_name); - } - } -} - -// Broadcast quantize info between the input and output of the OPs that will not -// change quantize info -void QuantizeModelProcessor::QuantizeInfoBroadcast() { - UpdateInputNameToNodes(); - for (auto iter = nodes_->begin(); iter < nodes_->end(); iter++) { - auto node = *iter; - if (node->op_type() != "Identity") { - continue; - } - std::string input_name = node->input(0); - std::string output_name = node->output(0); - auto input_quantize_info_iter = helper_->quantize_info.find(input_name); - auto output_quantize_info_iter = helper_->quantize_info.find(output_name); - // The input and output of Identity do not have quantize info - if (input_quantize_info_iter == helper_->quantize_info.end() && - output_quantize_info_iter == helper_->quantize_info.end()) { - continue; - } - // The input and output of Identity have quantize info - if (input_quantize_info_iter != helper_->quantize_info.end() && - output_quantize_info_iter != helper_->quantize_info.end()) { - continue; - } - if (input_quantize_info_iter != helper_->quantize_info.end()) { - helper_->quantize_info[output_name] = helper_->quantize_info[input_name]; - } else if (output_quantize_info_iter != helper_->quantize_info.end()) { - helper_->quantize_info[input_name] = helper_->quantize_info[output_name]; - } - if (ConnectToOutput(output_name)) { - continue; - } - RemoveNodeByName(node->name()); - iter--; - } -} - -bool QuantizeModelProcessor::IsGraphOutput(const std::string& name) { - for (auto& item : *outputs_) { - auto out_node = (*item.get()); - if (name == out_node.name()) { - return true; - } - } - return false; -} - -// Try get tensor shape value -bool QuantizeModelProcessor::GetTensorShape(const std::string& name, - std::vector* shape) { - for (auto& item : *parameters_) { - auto node = *(item.get()); - if (node.output(0) != name) { - continue; - } - for (auto i = 0; i < node.attribute_size(); i++) { - auto attr = node.attribute(i); - if (attr.name() == "value") { - auto tensor = attr.mutable_t(); - for (int64_t i = 0; i < tensor->dims_size(); i++) { - shape->push_back(tensor->dims(i)); - } - } - } - } - return !shape->empty(); -} - -void QuantizeModelProcessor::GetTensorWiseQuantizeInfo( - const std::vector& tensor, std::vector* scale, - std::vector* zero) { - float max_val = -1; - for (int64_t i = 0; i < tensor.size(); i++) { - if (fabs(tensor[i]) > max_val) { - max_val = fabs(tensor[i]); - } - } - Assert(max_val >= 0, - "[GetTensorWiseQuantizeInfo] Require the scale >= 0, but now it's " + - std::to_string(max_val) + "."); - scale->push_back(max_val / 127); - zero->push_back(0); -} - -void QuantizeModelProcessor::GetChannelWiseQuantizeInfo( - const std::vector& tensor, const std::vector& shape, - const int64_t& quant_axis, std::vector* scale, - std::vector* zero) { - int64_t channel_count = shape[quant_axis]; - - for (int64_t i = 0; i < channel_count; i++) { - if (quant_axis == 0) { - float max_val = -1; - int64_t inner_offset = 1; - for (auto& j : shape) { - inner_offset *= j; - } - inner_offset /= channel_count; - int64_t index = i * inner_offset; - for (int64_t j = 0; j < inner_offset; j++) { - if (fabs(tensor[index + j]) > max_val) { - max_val = fabs(tensor[index + j]); - } - } - Assert( - max_val >= 0, - "[GetChannelWiseQuantizeInfo] Require the scale >= 0, but now it's " + - std::to_string(max_val) + "."); - scale->push_back(max_val / 127); - zero->push_back(0); - } else if (quant_axis == 1) { - float max_val = -1; - int64_t inner_offset = shape.size() == 4 ? shape[2] * shape[3] : 1; - for (int64_t outter = 0; outter < shape[0]; outter++) { - int64_t index = outter * channel_count * inner_offset; - for (int64_t inner = 0; inner < inner_offset; inner++) { - int64_t final_index = index + i * inner_offset + inner; - if (fabs(tensor[final_index]) > max_val) { - max_val = fabs(tensor[final_index]); - } - } - } - Assert( - max_val >= 0, - "[GetChannelWiseQuantizeInfo] Require the scale >= 0, but now it's " + - std::to_string(max_val) + "."); - scale->push_back(max_val / 127); - zero->push_back(0); - } else { - Assert(false, - "QuantizeModelProcessor::GetChannelWiseQuantizeInfo only supports " - "quant_axis equals to 0 or 1, but now it's " + - std::to_string(quant_axis) + "."); - } - } -} - -template -bool QuantizeModelProcessor::GetTensorByName(const std::string& name, - std::vector* value) { - // Find tensor values in the following order, if found, store the data in - // value, and return true: - // 1. updated_parameters, the weight of conv or matmul. - // 2. parameters of original graph, the scale or bias of BN. - // 3. constant node in nodes, other vals. - auto updated_params_iter = helper_->updated_params.find(name); - if (updated_params_iter != helper_->updated_params.end()) { - (updated_params_iter->second).get(value); - return true; - } - for (int64_t block_index = 0; block_index < parser_->NumOfBlocks(); - block_index++) { - if (parser_->TryGetTensorValue(block_index, name, value)) { - return true; - } - } - return helper_->TryGetTensorValue(name, value); -} - -bool QuantizeModelProcessor::ConnectToOutput(const std::string& output_name) { - std::vector names = {output_name}; - while (!names.empty()) { - std::string name = names[names.size() - 1]; - names.pop_back(); - if (IsGraphOutput(name)) { - return true; - } - auto next_nodes = name2node_dict_[name]; - for (auto& next : next_nodes) { - if (next->op_type() == "Identity") { - names.push_back(next->output(0)); - } - } - } - return false; -} - -bool QuantizeModelProcessor::CanBeQuantize( - const std::vector& tensor_names, - const std::vector& output_index) { - for (auto& tensor : tensor_names) { - if (helper_->quantize_info.find(tensor) == helper_->quantize_info.end()) { - return false; - } - } - // If there is an OP linked to the output by identity, it needs to be skipped, - // do not quantize the OP - for (auto i = 0; i < output_index.size(); i++) { - int64_t index = output_index[i]; - if (index == -1) { - index = tensor_names.size() - 1; - } - - std::string output_name = tensor_names[index]; - if (ConnectToOutput(output_name)) { - return false; - } - } - return true; -} - -void QuantizeModelProcessor::AppendQuantizeTensor(const std::string& tensor, - const bool& only_dequantize) { - if (only_dequantize) { - if (std::find(only_dequantize_tensors.begin(), - only_dequantize_tensors.end(), - tensor) == only_dequantize_tensors.end()) { - only_dequantize_tensors.push_back(tensor); - } - } else { - if (std::find(tensors_to_be_quantize.begin(), tensors_to_be_quantize.end(), - tensor) == tensors_to_be_quantize.end()) { - tensors_to_be_quantize.push_back(tensor); - } - } -} -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/quantize_helper.h b/paddle2onnx/mapper/quantize_helper.h deleted file mode 100755 index 59d76d150bf..00000000000 --- a/paddle2onnx/mapper/quantize_helper.h +++ /dev/null @@ -1,135 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include - -#include -#include -#include - -#include "paddle2onnx/mapper/mapper.h" -#include "paddle2onnx/parser/parser.h" -namespace paddle2onnx { - -struct QuantizeModelProcessor { - public: - std::vector quantize_info; - const PaddleParser* parser_; - OnnxHelper* helper_; - - std::vector>* parameters_; - std::vector>* inputs_; - std::vector>* outputs_; - std::vector>* nodes_; - // All types that support quantization - std::vector supported_quantize_type_; - - std::map>> - name2node_dict_; - std::vector tensors_to_be_quantize; // records those tensors - // that need to add quantize - // and dequantize op - std::vector only_dequantize_tensors; // records those tensors - // that only need to add - // the dequantize op - // Convert to different model formats based on backend, backend can be - // TensorRT, ONNXRuntime and Others - void ProcessQuantizeModel( - std::vector>* parameters, - std::vector>* inputs, - std::vector>* outputs, - std::vector>* nodes, - OnnxHelper* helper, const std::string& deploy_backend, - const PaddleParser& parser, std::string* calibration_cache = nullptr); - - // Remove all Quantize and Dequantize ops - void RemoveAllQuantizeOps(); - - // If all tensors in tensor_names have quantize info and all the next nodes - // can be quantized, return True, otherwise - // return false - bool CanBeQuantize(const std::vector& tensor_names, - const std::vector& output_index = {-1}); - // only_dequantize records those tensors that only need to add the dequantize - // op - void AppendQuantizeTensor(const std::string& tensor, - const bool& only_dequantize = false); - - // Add QDQ for ORT according to: - // https://github.com/microsoft/onnxruntime/blob/master/onnxruntime/core/optimizer/qdq_transformer/selectors_actions/qdq_selector_action_transformer.cc - void AddQDQForORT(); - - // Determine if the tensor is directly linked to the output by identity - bool ConnectToOutput(const std::string& output_name); - - // Generate cache file for TensorRT8.X int8 deploy - void GenerateCache(std::string* calibration_cache); - - // Add QDQ for TRT according to: - // https://github.com/NVIDIA/TensorRT/tree/main/tools/pytorch-quantization/pytorch_quantization/nn/modules - void AddTrtQDQ(); - - // Add QDQ for RKNN - void AddQDQForRKNN(); - - void RemoveIdentityOp(); - - // Add quantize related op in model according to tensor names - void AddQDQInModel(const std::vector& tensors_to_be_quantize); - - void QuantizeInfoBroadcast(); - - // merge conv + add - void MergeConvAdd(); - - // merge conv + BN - void MergeConvBN(); - - // Determine whether a tensor is an output - bool IsGraphOutput(const std::string& name); - - // Because processing the quantize model will add new nodes, which will - // destroy the topo sorting of nodes, this function will sort the nodes again - void SortNodes(); - - bool GetTensorShape(const std::string& name, std::vector* shape); - - // return the value of tensor by name - template - bool GetTensorByName(const std::string& name, std::vector* value); - - // Perform tensor wise quantization, returning scale and zero - void GetTensorWiseQuantizeInfo(const std::vector& tensor, - std::vector* scale, - std::vector* zero); - - // Perform channel wise quantization, returning scale and zero - void GetChannelWiseQuantizeInfo(const std::vector& tensor, - const std::vector& shape, - const int64_t& quant_axis, - std::vector* scale, - std::vector* zero); - - // Generate name2node_dict to save input name and its related nodes - void UpdateInputNameToNodes(); - - void RemoveNodeByName(const std::string& name, const bool& update_io = true); - - void ReplaceInputOfAllNodes( - const std::string& old_name, const std::string& new_name, - const std::vector>& - except_nodes = {}); -}; -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/register_mapper.h b/paddle2onnx/mapper/register_mapper.h deleted file mode 100644 index 1d2253e0f31..00000000000 --- a/paddle2onnx/mapper/register_mapper.h +++ /dev/null @@ -1,115 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include -#include - -#include "paddle2onnx/utils/utils.h" -// This code is modified from -// https://blog.csdn.net/ZJU_fish1996/article/details/86515711 -namespace paddle2onnx { -class Mapper; -class PaddleParser; -class OnnxHelper; -#define REGISTER_MAPPER(op_name, class_name) \ - class op_name##Generator : public Generator { \ - public: \ - op_name##Generator() { MapperHelper::Get()->Push(#op_name, this); } \ - void Touch() {}; \ - Mapper* Create(const PaddleParser& p, OnnxHelper* h, int64_t b, \ - int64_t o) { \ - auto m = new class_name(p, h, b, o); \ - m->name_ = #class_name; \ - return m; \ - } \ - }; \ - op_name##Generator* op_name##inst = new op_name##Generator(); \ - int Touch##op_name##class_name() { \ - op_name##inst->Touch(); \ - return 0; \ - } - -class Generator { - public: - virtual Mapper* Create(const PaddleParser&, OnnxHelper* helper, int64_t, - int64_t) = 0; -}; - -class MapperHelper { - private: - std::map mappers; - std::map name_counter; - MapperHelper() {} - - public: - static MapperHelper* helper; - static MapperHelper* Get() { - if (nullptr == helper) { - helper = new MapperHelper(); - } - return helper; - } - - int64_t GetAllOps(const std::string& file_path) { - std::ofstream outfile(file_path); - if (!outfile) { - std::cerr << "Failed to open file: " << file_path << std::endl; - return mappers.size(); - } - for (auto iter = mappers.begin(); iter != mappers.end(); iter++) { - outfile << iter->first << std::endl; - } - outfile << "Total OPs: " << mappers.size() << std::endl; - std::cout << " [ * Paddle2ONNX * ] All Registered OPs saved in " - << file_path << std::endl; - outfile.close(); - return mappers.size(); - } - - bool IsRegistered(const std::string& op_name) { - auto iter = mappers.find(op_name); - if (mappers.end() == iter) { - return false; - } - return true; - } - - std::string GenName(const std::string& op_name) { - std::string key = "p2o." + op_name + "."; - if (name_counter.find(key) == name_counter.end()) { - name_counter[key] = 0; - } else { - name_counter[key] += 1; - } - return key + std::to_string(name_counter[key]); - } - - void ClearNameCounter() { name_counter.clear(); } - - Mapper* CreateMapper(const std::string& name, const PaddleParser& parser, - OnnxHelper* helper, int64_t block_id, int64_t op_id) { - Assert(mappers.find(name) != mappers.end(), - name + " cannot be found in registered mappers."); - return mappers[name]->Create(parser, helper, block_id, op_id); - } - - void Push(const std::string& name, Generator* generator) { - Assert(mappers.find(name) == mappers.end(), - name + " has been registered before."); - mappers[name] = generator; - } -}; -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/add_n.cc b/paddle2onnx/mapper/tensor/add_n.cc deleted file mode 100644 index 24c5c67da53..00000000000 --- a/paddle2onnx/mapper/tensor/add_n.cc +++ /dev/null @@ -1,38 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/add_n.h" - -namespace paddle2onnx { -REGISTER_MAPPER(sum, AddNMapper) - -void AddNMapper::Opset7() { - auto x_info = GetInput("X"); - auto out_info = GetOutput("Out"); - if (x_info.size() == 1) { - helper_->AutoCast(x_info[0].name, out_info[0].name, x_info[0].dtype, - out_info[0].dtype); - } else { - std::vector inputs; - for (auto i = 0; i < x_info.size(); ++i) { - inputs.push_back(helper_->AutoCast(x_info[i].name, x_info[0].dtype, - P2ODataType::FP32)); - } - auto output = helper_->MakeNode("Sum", inputs)->output(0); - helper_->AutoCast(output, out_info[0].name, P2ODataType::FP32, - out_info[0].dtype); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/add_n.h b/paddle2onnx/mapper/tensor/add_n.h deleted file mode 100644 index b74191a4642..00000000000 --- a/paddle2onnx/mapper/tensor/add_n.h +++ /dev/null @@ -1,31 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class AddNMapper : public Mapper { - public: - AddNMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/argmax.cc b/paddle2onnx/mapper/tensor/argmax.cc deleted file mode 100644 index f655865ec17..00000000000 --- a/paddle2onnx/mapper/tensor/argmax.cc +++ /dev/null @@ -1,70 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/argmax.h" - -namespace paddle2onnx { -REGISTER_MAPPER(arg_max, ArgMaxMapper) - -int32_t ArgMaxMapper::GetMinOpset(bool verbose) { - if (IsAttrVar("axis") && !IsConstant(GetAttrVar("axis")[0])) { - Error() << "While Attribute(axis)'s type is Tensor, it's not " - "supported " - "unless it's a constant tensor." - << std::endl; - return -1; - } - return 7; -} - -void ArgMaxMapper::Opset7() { - auto input_info = parser_->GetOpInput(block_idx_, op_idx_, "X"); - auto output_info = parser_->GetOpOutput(block_idx_, op_idx_, "Out"); - auto input = input_info[0].name; - if (flatten_) { - input = helper_->Flatten(input_info[0].name); - } - - if (IsAttrVar("axis")) { - auto axis_info = GetAttrVar("axis"); - std::vector temp; - TryGetValue(axis_info[0], &temp); - axis_ = temp[0]; - } else { - GetAttr("axis", &axis_); - } - if (input_info[0].dtype == P2ODataType::FP64) { - input = helper_->AutoCast(input, P2ODataType::FP64, P2ODataType::FP32); - } - if (input_info[0].dtype == P2ODataType::INT64) { - input = helper_->AutoCast(input, P2ODataType::INT64, P2ODataType::INT32); - } - auto arg_node = helper_->MakeNode("ArgMax", {input}); - AddAttribute(arg_node, "axis", axis_); - AddAttribute(arg_node, "keepdims", static_cast(keepdims_)); - if (keepdims_) { - std::vector shape(input_info[0].Rank(), 1); - std::string out = arg_node->output(0); - if (flatten_) { - out = helper_->Reshape(arg_node->output(0), shape); - } - helper_->AutoCast(out, output_info[0].name, P2ODataType::INT64, - output_info[0].dtype); - } else { - helper_->AutoCast(arg_node->output(0), output_info[0].name, - P2ODataType::INT64, output_info[0].dtype); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/argmax.h b/paddle2onnx/mapper/tensor/argmax.h deleted file mode 100755 index f7375e94e11..00000000000 --- a/paddle2onnx/mapper/tensor/argmax.h +++ /dev/null @@ -1,42 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class ArgMaxMapper : public Mapper { - public: - ArgMaxMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("flatten", &flatten_); - GetAttr("keepdims", &keepdims_); - GetAttr("dtype", &dtype_); - } - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - bool flatten_; - bool keepdims_; - int64_t axis_; - int64_t dtype_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/argmin.cc b/paddle2onnx/mapper/tensor/argmin.cc deleted file mode 100644 index 2b4136ae9d1..00000000000 --- a/paddle2onnx/mapper/tensor/argmin.cc +++ /dev/null @@ -1,71 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/argmin.h" - -namespace paddle2onnx { -REGISTER_MAPPER(arg_min, ArgMinMapper) - -int32_t ArgMinMapper::GetMinOpset(bool verbose) { - if (IsAttrVar("axis") && !IsConstant(GetAttrVar("axis")[0])) { - Error() << "While Attribute(axis)'s type is Tensor, it's not " - "supported " - "unless it's a constant tensor." - << std::endl; - return -1; - } - return 7; -} - -void ArgMinMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - auto input = input_info[0].name; - if (flatten_) { - input = helper_->Flatten(input_info[0].name); - } - - if (IsAttrVar("axis")) { - auto axis_info = GetAttrVar("axis"); - std::vector temp; - TryGetValue(axis_info[0], &temp); - axis_ = temp[0]; - } else { - GetAttr("axis", &axis_); - } - - if (input_info[0].dtype == P2ODataType::FP64) { - input = helper_->AutoCast(input, P2ODataType::FP64, P2ODataType::FP32); - } - if (input_info[0].dtype == P2ODataType::INT64) { - input = helper_->AutoCast(input, P2ODataType::INT64, P2ODataType::INT32); - } - auto arg_node = helper_->MakeNode("ArgMin", {input}); - AddAttribute(arg_node, "axis", axis_); - AddAttribute(arg_node, "keepdims", static_cast(keepdims_)); - if (keepdims_) { - std::vector shape(input_info[0].Rank(), 1); - std::string out = arg_node->output(0); - if (flatten_) { - out = helper_->Reshape(arg_node->output(0), shape); - } - helper_->AutoCast(out, output_info[0].name, P2ODataType::INT64, - output_info[0].dtype); - } else { - helper_->AutoCast(arg_node->output(0), output_info[0].name, - P2ODataType::INT64, output_info[0].dtype); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/argmin.h b/paddle2onnx/mapper/tensor/argmin.h deleted file mode 100755 index ca29d1a590c..00000000000 --- a/paddle2onnx/mapper/tensor/argmin.h +++ /dev/null @@ -1,42 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class ArgMinMapper : public Mapper { - public: - ArgMinMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("flatten", &flatten_); - GetAttr("keepdims", &keepdims_); - GetAttr("dtype", &dtype_); - } - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - bool flatten_; - bool keepdims_; - int64_t axis_; - int64_t dtype_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/argsort.cc b/paddle2onnx/mapper/tensor/argsort.cc deleted file mode 100644 index bef1002137f..00000000000 --- a/paddle2onnx/mapper/tensor/argsort.cc +++ /dev/null @@ -1,77 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/argsort.h" - -namespace paddle2onnx { -REGISTER_MAPPER(argsort, ArgsortMapper) - -int32_t ArgsortMapper::GetMinOpset(bool verbose) { - if (!descending_) { - Logger(verbose, 11) << "While descending=False, " << RequireOpset(11) - << std::endl; - return 11; - } - - if (axis_ < 0) { - axis_ = axis_ + GetInput("X")[0].Rank(); - } - if (GetInput("X")[0].shape[axis_] <= 0) { - Logger(verbose, 10) << "While input shape is dynamic, " << RequireOpset(10) - << std::endl; - return 10; - } - return 7; -} - -void ArgsortMapper::Opset10() { - auto x_info = GetInput("X"); - auto output_info = GetOutput("Out"); - auto indices_info = GetOutput("Indices"); - - auto shape = helper_->MakeNode("Shape", {x_info[0].name})->output(0); - if (axis_ < 0) { - axis_ = axis_ + x_info[0].Rank(); - } - auto dim_size = helper_->Slice(shape, {0}, {axis_}, {axis_ + 1}); - - auto out_node = - helper_->MakeNode("TopK", {x_info[0].name, dim_size}, - {output_info[0].name, indices_info[0].name}); - AddAttribute(out_node, "axis", axis_); - if (helper_->GetOpsetVersion() > 10) { - if (!descending_) { - AddAttribute(out_node, "largest", static_cast(0)); - } else { - AddAttribute(out_node, "largest", static_cast(1)); - } - } -} - -void ArgsortMapper::Opset7() { - auto x_info = GetInput("X"); - auto output_info = GetOutput("Out"); - auto indices_info = GetOutput("Indices"); - - if (axis_ < 0) { - axis_ = axis_ + x_info[0].Rank(); - } - - auto out_node = helper_->MakeNode( - "TopK", {x_info[0].name}, {output_info[0].name, indices_info[0].name}); - AddAttribute(out_node, "axis", axis_); - AddAttribute(out_node, "k", x_info[0].shape[axis_]); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/argsort.h b/paddle2onnx/mapper/tensor/argsort.h deleted file mode 100644 index c339566729e..00000000000 --- a/paddle2onnx/mapper/tensor/argsort.h +++ /dev/null @@ -1,37 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class ArgsortMapper : public Mapper { - public: - ArgsortMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("descending", &descending_); - GetAttr("axis", &axis_); - } - int32_t GetMinOpset(bool verbose = false); - void Opset10(); - void Opset7(); - - private: - bool descending_; - int64_t axis_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/assign.cc b/paddle2onnx/mapper/tensor/assign.cc deleted file mode 100644 index a4d0d3553de..00000000000 --- a/paddle2onnx/mapper/tensor/assign.cc +++ /dev/null @@ -1,48 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/assign.h" - -namespace paddle2onnx { -REGISTER_MAPPER(assign, AssignMapper) -REGISTER_MAPPER(share_data, AssignMapper) - -void AssignMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - if (block_idx_ != 0 && OpType() != "share_data") { - // Here's a trick for tensorrt - // Consider remove this trick - if (input_info[0].dtype == P2ODataType::BOOL) { - auto zero = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, - std::vector(1, 0)); - auto cast_input = helper_->AutoCast(input_info[0].name, P2ODataType::BOOL, - P2ODataType::INT64); - auto result = helper_->MakeNode("Add", {cast_input, zero})->output(0); - helper_->AutoCast(result, output_info[0].name, P2ODataType::INT64, - output_info[0].dtype); - } else { - auto zero = helper_->Constant(GetOnnxDtype(input_info[0].dtype), - std::vector(1, 0.0)); - auto new_input = - helper_->Unsqueeze(input_info[0].name, std::vector(1, 0)); - auto result = helper_->MakeNode("Add", {new_input, zero})->output(0); - helper_->Squeeze(result, output_info[0].name, std::vector(1, 0)); - } - } else { - helper_->MakeNode("Identity", {input_info[0].name}, {output_info[0].name}); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/assign.h b/paddle2onnx/mapper/tensor/assign.h deleted file mode 100644 index ad9585978f2..00000000000 --- a/paddle2onnx/mapper/tensor/assign.h +++ /dev/null @@ -1,31 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class AssignMapper : public Mapper { - public: - AssignMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/assign_value.cc b/paddle2onnx/mapper/tensor/assign_value.cc deleted file mode 100644 index 6f476cd4444..00000000000 --- a/paddle2onnx/mapper/tensor/assign_value.cc +++ /dev/null @@ -1,49 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/assign_value.h" - -#include -#include -#include - -namespace paddle2onnx { -REGISTER_MAPPER(assign_value, AssignValueMapper) - -int32_t AssignValueMapper::GetMinOpset(bool verbose) { - int32_t dtype = static_cast(dtype_); - if (dtype != P2ODataType::INT32 && dtype != P2ODataType::INT64 && - dtype != P2ODataType::FP32) { - Error() << "Only supports int32/int64/float32." << std::endl; - return -1; - } - return 7; -} - -void AssignValueMapper::Opset7() { - auto output_info = GetOutput("Out"); - int32_t dtype = static_cast(dtype_); - if (dtype == P2ODataType::INT32) { - helper_->Assign(output_info[0].name, GetOnnxDtype(output_info[0].dtype), - shape_, int64_values_); - } else if (dtype == P2ODataType::FP32) { - helper_->Assign(output_info[0].name, GetOnnxDtype(output_info[0].dtype), - shape_, fp32_values_); - } else if (dtype == P2ODataType::INT64) { - helper_->Assign(output_info[0].name, GetOnnxDtype(output_info[0].dtype), - shape_, int64_values_); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/assign_value.h b/paddle2onnx/mapper/tensor/assign_value.h deleted file mode 100644 index ebda2588571..00000000000 --- a/paddle2onnx/mapper/tensor/assign_value.h +++ /dev/null @@ -1,49 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class AssignValueMapper : public Mapper { - public: - AssignValueMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("dtype", &dtype_); - GetAttr("shape", &shape_); - int32_t dtype = static_cast(dtype_); - if (dtype == P2ODataType::INT32) { - GetAttr("int32_values", &int64_values_); - } else if (dtype == P2ODataType::FP32) { - GetAttr("fp32_values", &fp32_values_); - } else if (dtype == P2ODataType::INT64) { - GetAttr("int64_values", &int64_values_); - } - } - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - std::vector fp32_values_; - std::vector int64_values_; - std::vector shape_; - int64_t dtype_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/atan2.cc b/paddle2onnx/mapper/tensor/atan2.cc deleted file mode 100644 index e7c14ed8091..00000000000 --- a/paddle2onnx/mapper/tensor/atan2.cc +++ /dev/null @@ -1,76 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/atan2.h" -#define M_PI 3.14159265358979323846 /* pi */ - -namespace paddle2onnx { -REGISTER_MAPPER(atan2, Atan2Mapper) - -int32_t Atan2Mapper::GetMinOpset(bool verbose) { - if (GetInput("X1")[0].dtype == P2ODataType::INT32 || - GetInput("X2")[0].dtype == P2ODataType::INT32 || - GetInput("X1")[0].dtype == P2ODataType::INT64 || - GetInput("X2")[0].dtype == P2ODataType::INT64) { - Error() << "The input dtype should be float32 or float64. " << std::endl; - return -1; - } - Logger(verbose, 9) << RequireOpset(9) << std::endl; - return 9; -} - -void Atan2Mapper::Opset9() { - auto x_info = GetInput("X1"); - auto y_info = GetInput("X2"); - auto out_info = GetOutput("Out"); - - std::string input_x_name = x_info[0].name; - std::string input_y_name = y_info[0].name; - auto dtype = P2ODataType::FP32; - if (x_info[0].dtype == P2ODataType::FP64 || - y_info[0].dtype == P2ODataType::FP64) { - input_x_name = - helper_->AutoCast(x_info[0].name, x_info[0].dtype, P2ODataType::FP32); - input_y_name = - helper_->AutoCast(y_info[0].name, y_info[0].dtype, P2ODataType::FP32); - } - auto div = helper_->MakeNode("Div", {input_x_name, input_y_name}); - auto atan = helper_->MakeNode("Atan", {div->output(0)}); - - std::string zero_node = - helper_->Constant(GetOnnxDtype(dtype), std::vector{0.0}); - - auto minus_node = helper_->MakeNode("Less", {input_y_name, zero_node}); - - std::string condition_node = - helper_->AutoCast(minus_node->output(0), dtype, P2ODataType::BOOL); - - std::string pi_node = - helper_->Constant(GetOnnxDtype(dtype), std::vector{static_cast(M_PI)}); - - auto sign_node = helper_->MakeNode("Sign", {input_x_name}); - - auto mul_node = helper_->MakeNode("Mul", {sign_node->output(0), pi_node}); - - auto where_node = helper_->MakeNode( - "Where", {condition_node, mul_node->output(0), zero_node}); - - auto add_node = - helper_->MakeNode("Add", {atan->output(0), where_node->output(0)}); - - helper_->AutoCast(add_node->output(0), out_info[0].name, dtype, - out_info[0].dtype); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/atan2.h b/paddle2onnx/mapper/tensor/atan2.h deleted file mode 100644 index b56c1dc3976..00000000000 --- a/paddle2onnx/mapper/tensor/atan2.h +++ /dev/null @@ -1,29 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class Atan2Mapper : public Mapper { - public: - Atan2Mapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset9(); - int32_t GetMinOpset(bool verbose = false); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/bmm.cc b/paddle2onnx/mapper/tensor/bmm.cc deleted file mode 100644 index 7d8c0b76d2c..00000000000 --- a/paddle2onnx/mapper/tensor/bmm.cc +++ /dev/null @@ -1,29 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/bmm.h" - -namespace paddle2onnx { -REGISTER_MAPPER(bmm, BmmMapper) - -void BmmMapper::Opset7() { - auto x_info = GetInput("X"); - auto y_info = GetInput("Y"); - auto out_info = GetOutput("Out"); - - auto y = helper_->AutoCast(y_info[0].name, y_info[0].dtype, x_info[0].dtype); - helper_->MakeNode("MatMul", {x_info[0].name, y}, {out_info[0].name}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/bmm.h b/paddle2onnx/mapper/tensor/bmm.h deleted file mode 100644 index 0f0057f78de..00000000000 --- a/paddle2onnx/mapper/tensor/bmm.h +++ /dev/null @@ -1,31 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class BmmMapper : public Mapper { - public: - BmmMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/cast.cc b/paddle2onnx/mapper/tensor/cast.cc deleted file mode 100644 index 8e9c6c647df..00000000000 --- a/paddle2onnx/mapper/tensor/cast.cc +++ /dev/null @@ -1,28 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/cast.h" - -namespace paddle2onnx { -REGISTER_MAPPER(cast, CastMapper) - -void CastMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - auto node = - helper_->MakeNode("Cast", {input_info[0].name}, {output_info[0].name}); - AddAttribute(node, "to", GetOnnxDtype(out_dtype_)); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/cast.h b/paddle2onnx/mapper/tensor/cast.h deleted file mode 100644 index d95c9c6582f..00000000000 --- a/paddle2onnx/mapper/tensor/cast.h +++ /dev/null @@ -1,36 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class CastMapper : public Mapper { - public: - CastMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("out_dtype", &out_dtype_); - } - void Opset7(); - - private: - int64_t out_dtype_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/clip.cc b/paddle2onnx/mapper/tensor/clip.cc deleted file mode 100644 index 6ef10525f32..00000000000 --- a/paddle2onnx/mapper/tensor/clip.cc +++ /dev/null @@ -1,89 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/clip.h" - -namespace paddle2onnx { -REGISTER_MAPPER(clip, ClipMapper) - -int32_t ClipMapper::GetMinOpset(bool verbose) { - bool has_max_tensor_input = HasInput("Max"); - bool has_min_tensor_input = HasInput("Min"); - if (has_max_tensor_input || has_min_tensor_input) { - return 11; - } - return 7; -} - -void ClipMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - bool has_max_tensor_input = HasInput("Max"); - bool has_min_tensor_input = HasInput("Min"); - - if (has_max_tensor_input || has_min_tensor_input) { - bool dtype_converted = false; - std::string input_name = input_info[0].name; - int32_t dtype = input_info[0].dtype; - // onnxruntime only supports float input - if (input_info[0].dtype != P2ODataType::FP32) { - input_name = helper_->AutoCast(input_info[0].name, input_info[0].dtype, - P2ODataType::FP32); - dtype_converted = true; - dtype = P2ODataType::FP32; - } - std::string max_name; - if (has_max_tensor_input) { - auto max_info = GetInput("Max"); - max_name = helper_->AutoCast(max_info[0].name, max_info[0].dtype, dtype); - if (max_info[0].Rank() > 0) { - max_name = helper_->Squeeze(max_name, {}); - } - } else { - float max_val; - GetAttr("max", &max_val); - max_name = helper_->Constant({}, GetOnnxDtype(dtype), max_val); - } - std::string min_name; - if (has_min_tensor_input) { - auto min_info = GetInput("Min"); - min_name = helper_->AutoCast(min_info[0].name, min_info[0].dtype, dtype); - if (min_info[0].Rank() > 0) { - min_name = helper_->Squeeze(min_name, {}); - } - } else { - float min_val; - GetAttr("min", &min_val); - min_name = helper_->Constant({}, GetOnnxDtype(dtype), min_val); - } - if (dtype_converted) { - auto node = helper_->MakeNode("Clip", {input_name, min_name, max_name}); - helper_->AutoCast(node->output(0), output_info[0].name, P2ODataType::FP32, - output_info[0].dtype); - } else { - helper_->MakeNode("Clip", {input_name, min_name, max_name}, - {output_info[0].name}); - } - } else { - float max_val; - GetAttr("max", &max_val); - float min_val; - GetAttr("min", &min_val); - helper_->Clip(input_info[0].name, output_info[0].name, min_val, max_val, - input_info[0].dtype); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/clip.h b/paddle2onnx/mapper/tensor/clip.h deleted file mode 100644 index 2e2019da69e..00000000000 --- a/paddle2onnx/mapper/tensor/clip.h +++ /dev/null @@ -1,32 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class ClipMapper : public Mapper { - public: - ClipMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - int32_t GetMinOpset(bool verbose = false); - void Opset7(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/concat.cc b/paddle2onnx/mapper/tensor/concat.cc deleted file mode 100644 index 3b106bd5af2..00000000000 --- a/paddle2onnx/mapper/tensor/concat.cc +++ /dev/null @@ -1,72 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/concat.h" - -#include -#include - -namespace paddle2onnx { -REGISTER_MAPPER(concat, ConcatMapper) - -int32_t ConcatMapper::GetMinOpset(bool verbose) { - if (HasInput("AxisTensor") && !IsConstantInput("AxisTensor")) { - Error() << "While AxisTensor as input exists, it's not supported unless " - "it's a constant tensor." - << std::endl; - return -1; - } else if (IsAttrVar("axis") && !IsConstant(GetAttrVar("axis")[0])) { - Error() << "While Attribute(axis)'s type is Tensor, it's not supported " - "unless it's a constant tensor." - << std::endl; - return -1; - } - return 7; -} - -void ConcatMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - int32_t casted_dtype; - std::vector casted_names = - helper_->DtypeAlignment(input_info, &casted_dtype); - - auto parse_axis_value = [&](const TensorInfo& tensor_info, int64_t& axis) { - std::vector value; - TryGetValue(tensor_info, &value); - axis = value[0]; - }; - - bool has_axis_tensor_input = HasInput("AxisTensor"); - - int64_t axis = axis_; - // NOTE(Aurelius84): we need to deprecate this branch in the future. - if (has_axis_tensor_input) { - auto info = GetInput("AxisTensor"); - parse_axis_value(info[0], axis); - } else if (IsAttrVar("axis")) { - auto info = GetAttrVar("axis"); - parse_axis_value(info[0], axis); - } - if (axis < 0) { - axis = axis + input_info[0].Rank(); - } - auto node = helper_->MakeNode("Concat", casted_names); - AddAttribute(node, "axis", axis); - helper_->AutoCast(node->output(0), output_info[0].name, casted_dtype, - output_info[0].dtype); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/concat.h b/paddle2onnx/mapper/tensor/concat.h deleted file mode 100644 index 4a4a79df177..00000000000 --- a/paddle2onnx/mapper/tensor/concat.h +++ /dev/null @@ -1,37 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class ConcatMapper : public Mapper { - public: - ConcatMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("axis", &axis_); - } - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - int64_t axis_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/cumsum.cc b/paddle2onnx/mapper/tensor/cumsum.cc deleted file mode 100644 index f9861f3594b..00000000000 --- a/paddle2onnx/mapper/tensor/cumsum.cc +++ /dev/null @@ -1,57 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/cumsum.h" - -namespace paddle2onnx { -REGISTER_MAPPER(cumsum, CumsumMapper) - -int32_t CumsumMapper::GetMinOpset(bool verbose) { - Logger(verbose, 11) << RequireOpset(11) << std::endl; - return 11; -} - -void CumsumMapper::Opset11() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - if (input_info[0].Rank() == 0) { - auto axis_node = helper_->Constant({}, GetOnnxDtype(P2ODataType::INT64), 0); - auto unsqueeze_node = helper_->Unsqueeze(input_info[0].name, {0}); - auto cumsum_node = helper_->MakeNode("CumSum", {unsqueeze_node, axis_node}); - if (flatten_) { - helper_->AutoCast(cumsum_node->output(0), output_info[0].name, - input_info[0].dtype, output_info[0].dtype); - } else { - helper_->Squeeze(cumsum_node->output(0), output_info[0].name, {0}); - } - } else { - std::string axis_node; - if (IsAttrVar("axis")) { - auto axis_info = GetAttrVar("axis"); - axis_node = helper_->AutoCast(axis_info[0].name, axis_info[0].dtype, - P2ODataType::INT64); - } else { - GetAttr("axis", &axis_); - axis_node = - helper_->Constant({}, GetOnnxDtype(P2ODataType::INT64), axis_); - } - std::string input_node = input_info[0].name; - if (flatten_) { - input_node = helper_->Reshape(input_info[0].name, {-1}); - } - helper_->MakeNode("CumSum", {input_node, axis_node}, {output_info[0].name}); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/cumsum.h b/paddle2onnx/mapper/tensor/cumsum.h deleted file mode 100644 index 64c09d32ae2..00000000000 --- a/paddle2onnx/mapper/tensor/cumsum.h +++ /dev/null @@ -1,38 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class CumsumMapper : public Mapper { - public: - CumsumMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("flatten", &flatten_); - } - int32_t GetMinOpset(bool verbose = false); - void Opset11(); - - private: - int64_t axis_; - bool flatten_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/dist.cc b/paddle2onnx/mapper/tensor/dist.cc deleted file mode 100644 index b69bb3de707..00000000000 --- a/paddle2onnx/mapper/tensor/dist.cc +++ /dev/null @@ -1,67 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/dist.h" - -#include -#include - -namespace paddle2onnx { - -REGISTER_MAPPER(dist, DistMapper) - -const int g_NegIntInfinity = 0xFF800000; -const float g_NegFloatInfinity = *((float *)&g_NegIntInfinity); - -void DistMapper::Opset7() { - auto x_info = GetInput("X"); - auto y_info = GetInput("Y"); - auto output_info = GetOutput("Out"); - - auto sub_node = helper_->MakeNode("Sub", {x_info[0].name, y_info[0].name}); - auto abs_node = helper_->MakeNode("Abs", {sub_node->output(0)}); - - if (fabs(p_) < 1e-6) { - auto sign_node = helper_->MakeNode("Sign", {abs_node->output(0)}); - auto sum_node = helper_->MakeNode("ReduceSum", {sign_node->output(0)}); - AddAttribute(sum_node, "keepdims", static_cast(0)); - auto s_sum_node = helper_->Reshape(sum_node->output(0), {-1}); - helper_->AutoCast(s_sum_node, output_info[0].name, x_info[0].dtype, - output_info[0].dtype); - } else if (p_ == std::numeric_limits::infinity()) { - auto max_node = helper_->MakeNode("ReduceMax", {abs_node->output(0)}); - AddAttribute(max_node, "keepdims", static_cast(0)); - auto s_max_node = helper_->Reshape(max_node->output(0), {-1}); - helper_->AutoCast(s_max_node, output_info[0].name, x_info[0].dtype, - output_info[0].dtype); - } else if (p_ == g_NegFloatInfinity) { - auto min_node = helper_->MakeNode("ReduceMin", {abs_node->output(0)}); - AddAttribute(min_node, "keepdims", static_cast(0)); - auto s_min_node = helper_->Reshape(min_node->output(0), {-1}); - helper_->AutoCast(s_min_node, output_info[0].name, x_info[0].dtype, - output_info[0].dtype); - } else { - std::string p = helper_->Constant({1}, GetOnnxDtype(x_info[0].dtype), p_); - auto pow_node = helper_->MakeNode("Pow", {abs_node->output(0), p}); - - auto sum_node = helper_->MakeNode("ReduceSum", {pow_node->output(0)}); - AddAttribute(sum_node, "keepdims", static_cast(0)); - auto s_node = helper_->Reshape(sum_node->output(0), {-1}); - - auto p_1 = helper_->MakeNode("Reciprocal", {p}); - helper_->MakeNode("Pow", {s_node, p_1->output(0)}, {output_info[0].name}); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/dist.h b/paddle2onnx/mapper/tensor/dist.h deleted file mode 100644 index 8d7eeedaec5..00000000000 --- a/paddle2onnx/mapper/tensor/dist.h +++ /dev/null @@ -1,34 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class DistMapper : public Mapper { - public: - DistMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("p", &p_); - } - void Opset7(); - - private: - float p_; -}; -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/dot.cc b/paddle2onnx/mapper/tensor/dot.cc deleted file mode 100755 index 0349b80c3b7..00000000000 --- a/paddle2onnx/mapper/tensor/dot.cc +++ /dev/null @@ -1,41 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/dot.h" - -namespace paddle2onnx { -REGISTER_MAPPER(dot, DotMapper) - -void DotMapper::Opset7() { - auto input_x_info = GetInput("X"); - auto input_y_info = GetInput("Y"); - auto output_info = GetOutput("Out"); - - auto mul_node = - helper_->MakeNode("Mul", {input_x_info[0].name, input_y_info[0].name}); - - if (helper_->GetOpsetVersion() >= 13) { - std::string axes_node = helper_->Constant( - {1}, GetOnnxDtype(P2ODataType::INT64), input_x_info[0].Rank() - 1); - helper_->MakeNode("ReduceSum", {mul_node->output(0), axes_node}, - {output_info[0].name}); - } else { - auto reducesum_node = helper_->MakeNode("ReduceSum", {mul_node->output(0)}, - {output_info[0].name}); - std::vector axes = {input_x_info[0].Rank() - 1}; - AddAttribute(reducesum_node, "axes", axes); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/dot.h b/paddle2onnx/mapper/tensor/dot.h deleted file mode 100644 index a9575060fc9..00000000000 --- a/paddle2onnx/mapper/tensor/dot.h +++ /dev/null @@ -1,31 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class DotMapper : public Mapper { - public: - DotMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/equal.cc b/paddle2onnx/mapper/tensor/equal.cc deleted file mode 100644 index e06be2442ee..00000000000 --- a/paddle2onnx/mapper/tensor/equal.cc +++ /dev/null @@ -1,44 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/equal.h" - -namespace paddle2onnx { -REGISTER_MAPPER(equal, EqualMapper) - -int32_t EqualMapper::GetMinOpset(bool verbose) { - if (axis_ != -1) { - Error() << "axis attribute must be -1 in operator equal." << std::endl; - return -1; - } - return 7; -} - -void EqualMapper::Opset7() { - auto input_x_info = GetInput("X"); - auto input_y_info = GetInput("Y"); - auto output_info = GetOutput("Out"); - - std::string input_x = input_x_info[0].name; - std::string input_y = input_y_info[0].name; - if (helper_->GetOpsetVersion() < 11) { - input_x = helper_->AutoCast(input_x_info[0].name, input_x_info[0].dtype, - P2ODataType::INT32); - input_y = helper_->AutoCast(input_y_info[0].name, input_y_info[0].dtype, - P2ODataType::INT32); - } - helper_->MakeNode("Equal", {input_x, input_y}, {output_info[0].name}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/equal.h b/paddle2onnx/mapper/tensor/equal.h deleted file mode 100644 index 23a85b64934..00000000000 --- a/paddle2onnx/mapper/tensor/equal.h +++ /dev/null @@ -1,37 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class EqualMapper : public Mapper { - public: - EqualMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("axis", &axis_); - } - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - int64_t axis_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/expand.cc b/paddle2onnx/mapper/tensor/expand.cc deleted file mode 100644 index f1a9cbf5ea9..00000000000 --- a/paddle2onnx/mapper/tensor/expand.cc +++ /dev/null @@ -1,40 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/expand.h" - -namespace paddle2onnx { -REGISTER_MAPPER(expand, ExpandMapper) - -void ExpandMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - std::string expand_times = ""; - if (HasInput("expand_times_tensor")) { - auto info = GetInput("expand_times_tensor"); - expand_times = helper_->ConcatIndices(info); - } else if (HasInput("ExpandTimes")) { - auto info = GetInput("ExpandTimes"); - expand_times = helper_->AutoCast(info[0].name, info[0].dtype, P2ODataType::INT64); - } else { - expand_times = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, expand_times_); - } - - helper_->MakeNode("Tile", - {input_info[0].name, expand_times}, - {output_info[0].name}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/expand.h b/paddle2onnx/mapper/tensor/expand.h deleted file mode 100644 index 5d45aa6afee..00000000000 --- a/paddle2onnx/mapper/tensor/expand.h +++ /dev/null @@ -1,36 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class ExpandMapper : public Mapper { - public: - ExpandMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("expand_times", &expand_times_); - } - void Opset7(); - - private: - std::vector expand_times_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/expand_as.cc b/paddle2onnx/mapper/tensor/expand_as.cc deleted file mode 100644 index 5a678aff718..00000000000 --- a/paddle2onnx/mapper/tensor/expand_as.cc +++ /dev/null @@ -1,52 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/expand_as.h" - -#include -#include -#include - -namespace paddle2onnx { -REGISTER_MAPPER(expand_as_v2, ExpandAsMapper) - -int32_t ExpandAsMapper::GetMinOpset(bool verbose) { - if (target_shape_.size() == 0 && !HasInput("target_tensor")) { - Error() << "Attribute `target_shape` or input tensor `target_tensor` is " - "not exist" - << std::endl; - return -1; - } - Logger(verbose, 8) << RequireOpset(8) << std::endl; - return 8; -}; - -void ExpandAsMapper::Opset8() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - std::string target_shape = ""; - if (HasInput("target_tensor")) { - auto info = GetInput("target_tensor"); - target_shape = helper_->MakeNode("Shape", {info[0].name})->output(0); - } else { - target_shape = - helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, target_shape_); - } - - helper_->MakeNode("Expand", {input_info[0].name, target_shape}, - {output_info[0].name}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/expand_as.h b/paddle2onnx/mapper/tensor/expand_as.h deleted file mode 100644 index b45a4337864..00000000000 --- a/paddle2onnx/mapper/tensor/expand_as.h +++ /dev/null @@ -1,37 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class ExpandAsMapper : public Mapper { - public: - ExpandAsMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("target_shape", &target_shape_); - } - int32_t GetMinOpset(bool verbose = false); - void Opset8(); - - private: - std::vector target_shape_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/expand_v2.cc b/paddle2onnx/mapper/tensor/expand_v2.cc deleted file mode 100644 index c8758bbb03c..00000000000 --- a/paddle2onnx/mapper/tensor/expand_v2.cc +++ /dev/null @@ -1,61 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/expand_v2.h" - -namespace paddle2onnx { -REGISTER_MAPPER(expand_v2, ExpandV2Mapper) - -void ExpandV2Mapper::Opset8() { - auto x_info = GetInput("X"); - auto out_info = GetOutput("Out"); - - int dim_diff = 0; - std::string shape = ""; - if (HasInput("Shape")) { - auto shape_info = GetInput("Shape"); - dim_diff = shape_info[0].shape[0] - x_info[0].Rank(); - shape = helper_->AutoCast(shape_info[0].name, shape_info[0].dtype, - P2ODataType::INT64); - } else if (HasInput("expand_shapes_tensor")) { - auto shape_info = GetInput("expand_shapes_tensor"); - dim_diff = shape_info.size() - x_info[0].Rank(); - shape = helper_->ConcatIndices(shape_info); - } else { - std::vector shape_value; - GetAttr("shape", &shape_value); - dim_diff = shape_value.size() - x_info[0].Rank(); - shape = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, shape_value); - } - - auto input_shape = helper_->MakeNode("Shape", {x_info[0].name})->output(0); - if (dim_diff > 0) { - auto padding_shape = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, - std::vector(dim_diff, 1)); - input_shape = helper_->Concat({padding_shape, input_shape}, 0); - } - if (helper_->GetOpsetVersion() < 12) { - // While opset < 12, Max cannot support int64 datatype with onnxruntime - input_shape = - helper_->AutoCast(input_shape, P2ODataType::INT64, P2ODataType::FP32); - shape = helper_->AutoCast(shape, P2ODataType::INT64, P2ODataType::FP32); - shape = helper_->MakeNode("Max", {input_shape, shape})->output(0); - shape = helper_->AutoCast(shape, P2ODataType::FP32, P2ODataType::INT64); - } else { - shape = helper_->MakeNode("Max", {input_shape, shape})->output(0); - } - helper_->MakeNode("Expand", {x_info[0].name, shape}, {out_info[0].name}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/expand_v2.h b/paddle2onnx/mapper/tensor/expand_v2.h deleted file mode 100644 index 00dc098c8ff..00000000000 --- a/paddle2onnx/mapper/tensor/expand_v2.h +++ /dev/null @@ -1,32 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class ExpandV2Mapper : public Mapper { - public: - ExpandV2Mapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - int32_t GetMinOpset(bool verbose = false) { - Logger(verbose, 8) << RequireOpset(8) << std::endl; - return 8; - } - void Opset8(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/eye.cc b/paddle2onnx/mapper/tensor/eye.cc deleted file mode 100644 index 5af3aa67577..00000000000 --- a/paddle2onnx/mapper/tensor/eye.cc +++ /dev/null @@ -1,87 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/eye.h" - -namespace paddle2onnx { -REGISTER_MAPPER(eye, EyeMapper) - -void EyeMapper::ParseValue(const TensorInfo& tensor_info, int64_t* num_val) { - std::vector value; - TryGetValue(tensor_info, &value); - *num_val = value[0]; -} - -int32_t EyeMapper::GetMinOpset(bool verbose) { - if (IsAttrVar("num_rows")) { - if (!IsConstant(GetAttrVar("num_rows")[0])) { - Error() - << "While Attribute(num_rows)'s type is Tensor, it's not supported " - "unless it's a constant tensor." - << std::endl; - return -1; - } else { - auto info = GetAttrVar("num_rows"); - ParseValue(info[0], &num_rows_); - } - } else { - GetAttr("num_rows", &num_rows_); - } - if (IsAttrVar("num_columns")) { - if (!IsConstant(GetAttrVar("num_columns")[0])) { - Error() << "While Attribute(num_columns)'s type is Tensor, it's not " - "supported " - "unless it's a constant tensor." - << std::endl; - return -1; - } else { - auto info = GetAttrVar("num_columns"); - ParseValue(info[0], &num_columns_); - } - } else { - GetAttr("num_columns", &num_columns_); - } - - if (num_rows_ <= 0 || num_columns_ <= 0) { - Error() << "Attribute `num_rows` or `num_columns` must greater than 0. " - << std::endl; - return -1; - } - Logger(verbose, 9) << RequireOpset(9) << std::endl; - return 9; -} - -void EyeMapper::Opset9() { - auto output_info = GetOutput("Out"); - if (IsAttrVar("num_rows")) { - auto info = GetAttrVar("num_rows"); - ParseValue(info[0], &num_rows_); - } else { - GetAttr("num_rows", &num_rows_); - } - if (IsAttrVar("num_columns")) { - auto info = GetAttrVar("num_columns"); - ParseValue(info[0], &num_columns_); - } else { - GetAttr("num_columns", &num_columns_); - } - - std::string constant_node = helper_->Constant( - {num_rows_, num_columns_}, GetOnnxDtype(output_info[0].dtype), 0); - - auto node = - helper_->MakeNode("EyeLike", {constant_node}, {output_info[0].name}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/eye.h b/paddle2onnx/mapper/tensor/eye.h deleted file mode 100644 index 7aa8629b281..00000000000 --- a/paddle2onnx/mapper/tensor/eye.h +++ /dev/null @@ -1,37 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class EyeMapper : public Mapper { - public: - EyeMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - int32_t GetMinOpset(bool verbose = false); - void Opset9(); - - private: - void ParseValue(const TensorInfo& tensor_info, int64_t* num_val); - int64_t num_rows_; - int64_t num_columns_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/fill_constant.cc b/paddle2onnx/mapper/tensor/fill_constant.cc deleted file mode 100644 index 13b68b52869..00000000000 --- a/paddle2onnx/mapper/tensor/fill_constant.cc +++ /dev/null @@ -1,153 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. -#include "paddle2onnx/mapper/tensor/fill_constant.h" - -#include -#include - -namespace paddle2onnx { - -REGISTER_MAPPER(fill_constant, FillConstantMapper) - -int32_t FillConstantMapper::GetMinOpset(bool verbose) { - auto out_info = GetOutput("Out"); - auto onnx_dtype = GetOnnxDtype(out_info[0].dtype); - if (onnx_dtype != ONNX_NAMESPACE::TensorProto::INT32 && - onnx_dtype != ONNX_NAMESPACE::TensorProto::INT64 && - onnx_dtype != ONNX_NAMESPACE::TensorProto::FLOAT && - onnx_dtype != ONNX_NAMESPACE::TensorProto::DOUBLE) { - Error() << "Only support int32/int64/float32/float64 data type in " - "fill_constant operator." - << std::endl; - return -1; - } - if (HasInput("ShapeTensorList")) { - Logger(verbose, 9) << "While ShapeTensorList as input, " << RequireOpset(9) << std::endl; - return 9; - } - if (HasInput("ShapeTensor") && !IsConstantInput("ShapeTensor")) { - Logger(verbose, 9) << "While ShapeTensor as input and it's not a constant tensor, " << RequireOpset(9) << std::endl; - return 9; - } - return 7; -} - -float FillConstantMapper::GetFillValue() { - float value = 0; - if (str_value_.empty()) { - value = value_; - } else { - if (str_value_ == "inf") { - value = std::numeric_limits::infinity(); - } else if (str_value_ == "-inf") { - value = -std::numeric_limits::infinity(); - } else if (str_value_ == "nan") { - value = std::numeric_limits::quiet_NaN(); - } else { - std::stringstream convert_stream(str_value_); - convert_stream >> value; - } - } - if (HasInput("ValueTensor")) { - value = 0.0; - } - return value; -} - -void FillConstantMapper::Opset7() { - auto out_info = GetOutput("Out"); - Assert(!HasInput("ShapeTensorList"), "While ShapeTensorList as input, requires opset_version>=9 for op fill_constant."); - std::vector shape; - if (HasInput("ShapeTensor")) { - Assert(TryGetInputValue("ShapeTensor", &shape), "While ShapeTensor as input and it's not a constant tensor, requires opset_version>=9 for op fill_constant."); - } else { - GetAttr("shape", &shape); - } - float value = GetFillValue(); - if (HasInput("ValueTensor")) { - auto value_info = GetInput("ValueTensor"); - auto value_tensor = helper_->AutoCast(value_info[0].name, value_info[0].dtype, out_info[0].dtype); - auto out = helper_->Constant(shape, GetOnnxDtype(out_info[0].dtype), float(0.0)); - helper_->MakeNode("Add", {out, value_tensor}, {out_info[0].name}); - } else { - helper_->Constant(out_info[0].name, shape, GetOnnxDtype(out_info[0].dtype), value); - } -} - -void FillConstantMapper::Opset9() { - if (GetMinOpset() == 7) { - return Opset7(); - } - auto out_info = GetOutput("Out"); - bool shape_is_tensor = HasInput("ShapeTensor") || HasInput("ShapeTensorList"); - bool value_is_tensor = HasInput("ValueTensor"); - auto onnx_dtype = GetOnnxDtype(out_info[0].dtype); - float value = GetFillValue(); - std::string out; - if (shape_is_tensor) { - std::string shape_name; - if (HasInput("ShapeTensor")) { - auto shape_info = GetInput("ShapeTensor"); - shape_name = helper_->AutoCast(shape_info[0].name, shape_info[0].dtype, - P2ODataType::INT64); - } else { - auto shape_info = GetInput("ShapeTensorList"); - shape_name = helper_->ConcatIndices(shape_info); - } - - auto node = helper_->MakeNode("ConstantOfShape", {shape_name}); - auto attr = node->add_attribute(); - attr->set_name("value"); - attr->set_type(ONNX_NAMESPACE::AttributeProto::TENSOR); - auto tensor = attr->mutable_t(); - tensor->set_name(out_info[0].name); - tensor->set_data_type(onnx_dtype); - tensor->add_dims(1); - if (onnx_dtype == ONNX_NAMESPACE::TensorProto::INT32) { - std::vector data(1); - data[0] = static_cast(value); - auto ptr = reinterpret_cast(data.data()); - tensor->set_raw_data(std::string(ptr, sizeof(int32_t))); - } else if (onnx_dtype == ONNX_NAMESPACE::TensorProto::INT64) { - std::vector data(1); - data[0] = static_cast(value); - auto ptr = reinterpret_cast(data.data()); - tensor->set_raw_data(std::string(ptr, sizeof(int64_t))); - } else if (onnx_dtype == ONNX_NAMESPACE::TensorProto::FLOAT) { - std::vector data(1, value_); - auto ptr = reinterpret_cast(data.data()); - tensor->set_raw_data(std::string(ptr, sizeof(float))); - } else if (onnx_dtype == ONNX_NAMESPACE::TensorProto::DOUBLE) { - std::vector data(1); - data[0] = static_cast(value); - auto ptr = reinterpret_cast(data.data()); - tensor->set_raw_data(std::string(ptr, sizeof(double))); - } - out = node->output(0); - } else { - std::vector shape; - GetAttr("shape", &shape); - out = helper_->Constant(shape, onnx_dtype, value); - } - if (value_is_tensor) { - auto value_info = GetInput("ValueTensor"); - std::string cast_value = helper_->AutoCast( - value_info[0].name, value_info[0].dtype, out_info[0].dtype); - helper_->MakeNode("Add", {out, cast_value}, {out_info[0].name}); - } else { - helper_->MakeNode("Identity", {out}, {out_info[0].name}); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/fill_constant.h b/paddle2onnx/mapper/tensor/fill_constant.h deleted file mode 100644 index 231cb840215..00000000000 --- a/paddle2onnx/mapper/tensor/fill_constant.h +++ /dev/null @@ -1,38 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. -#pragma once -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class FillConstantMapper : public Mapper { - public: - FillConstantMapper(const PaddleParser& p, OnnxHelper* helper, - int64_t block_id, int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("str_value", &str_value_); - GetAttr("value", &value_); - } - - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - void Opset9(); - - private: - float GetFillValue(); - std::string str_value_; - float value_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/fill_constant_batch_size_like.cc b/paddle2onnx/mapper/tensor/fill_constant_batch_size_like.cc deleted file mode 100644 index 80fbc787727..00000000000 --- a/paddle2onnx/mapper/tensor/fill_constant_batch_size_like.cc +++ /dev/null @@ -1,84 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. -#include "paddle2onnx/mapper/tensor/fill_constant_batch_size_like.h" - -namespace paddle2onnx { - -REGISTER_MAPPER(fill_constant_batch_size_like, FillConstantBatchSizeLikeMapper) - -int32_t FillConstantBatchSizeLikeMapper::GetMinOpset(bool verbose) { - auto out_info = GetOutput("Out"); - if (out_info[0].dtype == P2ODataType::BOOL) { - Error() << "Dtype of boolean is not supported." << std::endl; - return -1; - } - return 7; -} - -void FillConstantBatchSizeLikeMapper::Opset7() { - auto input_info = GetInput("Input"); - auto out_info = GetOutput("Out"); - float value = value_; - if (!str_value_.empty()) { - std::stringstream convert_stream(str_value_); - convert_stream >> value; - } - std::vector shape; - shape.assign(shape_.begin(), shape_.end()); - shape[output_dim_idx_] = 1; - - auto input_shape = - helper_->MakeNode("Shape", {input_info[0].name})->output(0); - auto batch = - helper_->Slice(input_shape, {0}, {input_dim_idx_}, {input_dim_idx_ + 1}); - - if (output_dim_idx_ == 0) { - auto constant = - helper_->Constant(shape, GetOnnxDtype(out_info[0].dtype), value); - auto repeat = batch; - if (shape.size() > 1) { - auto tmp = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, - std::vector(shape.size() - 1, 1)); - repeat = helper_->Concat({batch, tmp}, 0); - } - helper_->MakeNode("Tile", {constant, repeat}, {out_info[0].name}); - } else if (output_dim_idx_ == shape.size() - 1) { - auto constant = - helper_->Constant(shape, GetOnnxDtype(out_info[0].dtype), value); - auto repeat = batch; - if (shape.size() > 1) { - auto tmp = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, - std::vector(shape.size() - 1, 1)); - repeat = helper_->Concat({tmp, batch}, 0); - } - helper_->MakeNode("Tile", {constant, repeat}, {out_info[0].name}); - } else { - shape.erase(shape.begin() + output_dim_idx_); - shape.insert(shape.begin(), int64_t(1)); - auto constant = - helper_->Constant(shape, GetOnnxDtype(out_info[0].dtype), value); - auto repeat = batch; - if (shape.size() > 1) { - auto tmp = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, - std::vector(shape.size() - 1, 1)); - repeat = helper_->Concat({batch, tmp}, 0); - } - auto out = helper_->MakeNode("Tile", {constant, repeat})->output(0); - auto perm = Arange(1, shape.size()); - perm.insert(perm.begin() + output_dim_idx_, int64_t(0)); - helper_->Transpose(out, out_info[0].name, perm); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/fill_constant_batch_size_like.h b/paddle2onnx/mapper/tensor/fill_constant_batch_size_like.h deleted file mode 100644 index c58ecf870dd..00000000000 --- a/paddle2onnx/mapper/tensor/fill_constant_batch_size_like.h +++ /dev/null @@ -1,48 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. -#pragma once -#include -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class FillConstantBatchSizeLikeMapper : public Mapper { - public: - FillConstantBatchSizeLikeMapper(const PaddleParser& p, OnnxHelper* helper, - int64_t block_id, int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("dtype", &dtype_); - GetAttr("value", &value_); - GetAttr("shape", &shape_); - GetAttr("str_value", &str_value_); - GetAttr("input_dim_idx", &input_dim_idx_); - GetAttr("output_dim_idx", &output_dim_idx_); - } - - int32_t GetMinOpset(bool verbose = true); - void Opset7(); - - private: - int64_t dtype_; - float value_; - std::string str_value_; - int64_t input_dim_idx_; - int64_t output_dim_idx_; - std::vector shape_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/fill_like.cc b/paddle2onnx/mapper/tensor/fill_like.cc deleted file mode 100644 index 5759502432f..00000000000 --- a/paddle2onnx/mapper/tensor/fill_like.cc +++ /dev/null @@ -1,48 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. -#include "paddle2onnx/mapper/tensor/fill_like.h" - -namespace paddle2onnx { - -REGISTER_MAPPER(fill_any_like, FillLikeMapper) -REGISTER_MAPPER(fill_zeros_like, FillLikeMapper) - -void FillLikeMapper::Opset9() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - bool is_fixed_shape = true; - for (size_t i = 0; i < input_info[0].shape.size(); ++i) { - if (input_info[0].shape[i] < 0) { - is_fixed_shape = false; - } - } - if (is_fixed_shape) { - helper_->Constant(output_info[0].name, input_info[0].shape, - GetOnnxDtype(output_info[0].dtype), value_); - return; - } - auto shape_node = helper_->MakeNode("Shape", {input_info[0].name}); - int64_t dtype = output_info[0].dtype; - // There's some problem with tensorrt with `ConstantOfShape` - // Maybe we should use a graph pass to solve this problem - // but now we just to avoid using `ConstantOfShape` - auto const_node = helper_->Constant({1}, GetOnnxDtype(dtype), value_); - helper_->MakeNode("Expand", {const_node, shape_node->output(0)}, - {output_info[0].name}); - // helper_->ConstOfShape(shape_node->output(0), output_info[0].name, - // GetOnnxDtype(dtype), value_); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/fill_like.h b/paddle2onnx/mapper/tensor/fill_like.h deleted file mode 100644 index b0b0ad23970..00000000000 --- a/paddle2onnx/mapper/tensor/fill_like.h +++ /dev/null @@ -1,46 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. -#pragma once -#include -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class FillLikeMapper : public Mapper { - public: - FillLikeMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - if (OpType() == "fill_zeros_like") { - value_ = 0.0; - } else { - GetAttr("value", &value_); - } - } - - int32_t GetMinOpset(bool verbose = false) { - Logger(verbose, 9) << RequireOpset(9) << std::endl; - return 9; - } - void Opset9(); - - private: - float value_; - std::vector op_mapper_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/flatten.cc b/paddle2onnx/mapper/tensor/flatten.cc deleted file mode 100644 index 5407138f497..00000000000 --- a/paddle2onnx/mapper/tensor/flatten.cc +++ /dev/null @@ -1,75 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/flatten.h" - -#include - -namespace paddle2onnx { - -REGISTER_MAPPER(flatten_contiguous_range, FlattenMapper) - -void FlattenMapper::Opset7() { - auto input_info = GetInput("X"); - if (start_axis_ < 0) { - start_axis_ += input_info[0].Rank(); - } - if (stop_axis_ < 0) { - stop_axis_ += input_info[0].Rank(); - } - auto output_info = GetOutput("Out"); - - auto unknown_dim_node = - helper_->Constant({1}, ONNX_NAMESPACE::TensorProto::INT64, -1); - if (start_axis_ == 0 && stop_axis_ == input_info[0].Rank() - 1) { - helper_->MakeNode("Reshape", {input_info[0].name, unknown_dim_node}, - {output_info[0].name}); - } else { - auto input_shape_node = helper_->MakeNode("Shape", {input_info[0].name}); - if (start_axis_ == 0) { - auto second_part_shape = - helper_->Slice(input_shape_node->output(0), {0}, {stop_axis_ + 1}, - {input_info[0].Rank()}); - auto new_shape_node = - helper_->MakeNode("Concat", {unknown_dim_node, second_part_shape}); - AddAttribute(new_shape_node, "axis", int64_t(0)); - helper_->MakeNode("Reshape", - {input_info[0].name, new_shape_node->output(0)}, - {output_info[0].name}); - } else if (stop_axis_ == input_info[0].Rank() - 1) { - auto first_part_shape = - helper_->Slice(input_shape_node->output(0), {0}, {0}, {start_axis_}); - auto new_shape_node = - helper_->MakeNode("Concat", {first_part_shape, unknown_dim_node}); - AddAttribute(new_shape_node, "axis", int64_t(0)); - helper_->MakeNode("Reshape", - {input_info[0].name, new_shape_node->output(0)}, - {output_info[0].name}); - } else { - auto first_part_shape = - helper_->Slice(input_shape_node->output(0), {0}, {0}, {start_axis_}); - auto second_part_shape = - helper_->Slice(input_shape_node->output(0), {0}, {stop_axis_ + 1}, - {input_info[0].Rank()}); - auto new_shape_node = helper_->MakeNode( - "Concat", {first_part_shape, unknown_dim_node, second_part_shape}); - AddAttribute(new_shape_node, "axis", int64_t(0)); - helper_->MakeNode("Reshape", - {input_info[0].name, new_shape_node->output(0)}, - {output_info[0].name}); - } - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/flatten.h b/paddle2onnx/mapper/tensor/flatten.h deleted file mode 100644 index 0789e49bf73..00000000000 --- a/paddle2onnx/mapper/tensor/flatten.h +++ /dev/null @@ -1,37 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class FlattenMapper : public Mapper { - public: - FlattenMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("start_axis", &start_axis_); - GetAttr("stop_axis", &stop_axis_); - } - - void Opset7(); - - private: - int64_t start_axis_; - int64_t stop_axis_; -}; -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/flatten2.cc b/paddle2onnx/mapper/tensor/flatten2.cc deleted file mode 100644 index 2615efa1e32..00000000000 --- a/paddle2onnx/mapper/tensor/flatten2.cc +++ /dev/null @@ -1,40 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/flatten2.h" - -#include -#include - -namespace paddle2onnx { -REGISTER_MAPPER(flatten2, Flatten2Mapper) - -int32_t Flatten2Mapper::GetMinOpset(bool verbose) { - if (GetInput("X")[0].dtype != P2ODataType::FP32 || GetInput("X")[0].dtype != P2ODataType::FP64) { - Logger(verbose, 9) << "While data type of input is not float32/float64, "<< RequireOpset(9) << std::endl; - return 9; - } - return 7; -} - -void Flatten2Mapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - int64_t axis = axis_; - auto node = helper_->MakeNode("Flatten", {input_info[0].name}, {output_info[0].name}); - AddAttribute(node, "axis", axis); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/flatten2.h b/paddle2onnx/mapper/tensor/flatten2.h deleted file mode 100644 index 05656335d30..00000000000 --- a/paddle2onnx/mapper/tensor/flatten2.h +++ /dev/null @@ -1,37 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class Flatten2Mapper : public Mapper { - public: - Flatten2Mapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("axis", &axis_); - } - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - int64_t axis_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/flip.cc b/paddle2onnx/mapper/tensor/flip.cc deleted file mode 100644 index b7bfe07c967..00000000000 --- a/paddle2onnx/mapper/tensor/flip.cc +++ /dev/null @@ -1,91 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/flip.h" - -namespace paddle2onnx { -REGISTER_MAPPER(flip, FlipMapper) - -int32_t FlipMapper::GetMinOpset(bool verbose) { - auto input_info = parser_->GetOpInput(block_idx_, op_idx_, "X"); - for (auto i = 0; i < axes_.size(); i++) { - if (input_info[0].shape[axes_[i]] <= 0) { - Error() << "The dimension in axis of input must be fixed for flip " - "operator, but now the input shape in axis is unkown." - << std::endl; - return -1; - } - } - return 7; -} - -void FlipMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - if (input_info[0].Rank() == 0) { - helper_->MakeNode("Identity", {input_info[0].name}, {output_info[0].name}); - } - std::string input_name = input_info[0].name; - bool need_convert = false; - if (input_info[0].dtype == P2ODataType::BOOL || - input_info[0].dtype == P2ODataType::FP64) { - need_convert = true; - input_name = helper_->AutoCast(input_info[0].name, input_info[0].dtype, - P2ODataType::FP32); - } - - std::string temp_input = input_name; - for (auto i = 0; i < axes_.size(); ++i) { - int64_t axis = axes_[i]; - if (input_info[0].shape[axis] == 1) { - if (i != axes_.size() - 1) { - continue; - } - if (need_convert) { - input_name = helper_->AutoCast(temp_input, output_info[0].name, - P2ODataType::FP32, output_info[0].dtype); - } else { - auto out_node = - helper_->MakeNode("Identity", {temp_input}, {output_info[0].name}); - } - } else { - std::vector split; - split.resize(input_info[0].shape[axis], 1); - std::vector splits_outputs = - helper_->Split(temp_input, split, axis); - std::vector reversed_splits; - for (int64_t index = splits_outputs.size() - 1; index >= 0; --index) { - reversed_splits.push_back(splits_outputs[index]); - } - if (i != axes_.size() - 1) { - auto concat_node = helper_->MakeNode("Concat", reversed_splits); - AddAttribute(concat_node, "axis", axis); - temp_input = concat_node->output(0); - } else { - if (need_convert) { - auto concat_node = helper_->MakeNode("Concat", reversed_splits); - AddAttribute(concat_node, "axis", axis); - helper_->AutoCast(concat_node->output(0), output_info[0].name, - P2ODataType::FP32, output_info[0].dtype); - } else { - auto concat_node = helper_->MakeNode("Concat", reversed_splits, - {output_info[0].name}); - AddAttribute(concat_node, "axis", axis); - } - } - } - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/flip.h b/paddle2onnx/mapper/tensor/flip.h deleted file mode 100755 index 5e40d3e1aab..00000000000 --- a/paddle2onnx/mapper/tensor/flip.h +++ /dev/null @@ -1,40 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class FlipMapper : public Mapper { - public: - FlipMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("axis", &axes_); - auto input_info = GetInput("X"); - for (auto i = 0; i < axes_.size(); i++) { - if (axes_[i] < 0) { - axes_[i] += input_info[0].Rank(); - } - } - } - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - std::vector axes_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/gather.cc b/paddle2onnx/mapper/tensor/gather.cc deleted file mode 100644 index 4cfb6f371ac..00000000000 --- a/paddle2onnx/mapper/tensor/gather.cc +++ /dev/null @@ -1,85 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/gather.h" - -namespace paddle2onnx { -REGISTER_MAPPER(gather, GatherMapper) - -int32_t GatherMapper::GetMinOpset(bool verbose) { - if (HasInput("Axis")) { - if (!IsConstantInput("Axis")) { - Error() << "Parameter axis as input tensor is not supported." - << std::endl; - return -1; - } - } - auto index_info = GetInput("Index"); - if (index_info[0].shape.size() > 1) { - Logger(verbose, 11) << "While rank of index > 1, " << RequireOpset(11) - << std::endl; - return 11; - } - return 7; -} - -void GatherMapper::Opset7() { - auto x_info = GetInput("X"); - auto index_info = GetInput("Index"); - auto out_info = GetOutput("Out"); - - bool has_input_axis = HasInput("Axis"); - auto axis = axis_; - if (has_input_axis) { - std::vector axes; - Assert(TryGetInputValue("Axis", &axes), - "Paddle2ONNX does not support axis as input tensor for operator: " - "gather."); - axis = axes[0]; - } - Assert(index_info[0].shape.size() == 1, - "Paddle2ONNX: While rank of index > 1, opset must >= 11 for operator: " - "gather."); - auto node = helper_->MakeNode("Gather", {x_info[0].name, index_info[0].name}, - {out_info[0].name}); - AddAttribute(node, "axis", axis); -} - -void GatherMapper::Opset11() { - auto x_info = GetInput("X"); - auto index_info = GetInput("Index"); - auto out_info = GetOutput("Out"); - - bool has_input_axis = HasInput("Axis"); - auto axis = axis_; - if (has_input_axis) { - std::vector axes; - Assert(TryGetInputValue("Axis", &axes), - "Paddle2ONNX does not support axis as input tensor for operator: " - "gather."); - axis = axes[0]; - } - if (index_info[0].shape.size() == 1) { - auto node = helper_->MakeNode( - "Gather", {x_info[0].name, index_info[0].name}, {out_info[0].name}); - AddAttribute(node, "axis", axis); - } else { - auto index = helper_->AutoCast(index_info[0].name, index_info[0].dtype, - P2ODataType::INT64); - helper_->MakeNode("GatherND", {x_info[0].name, index_info[0].name}, - {out_info[0].name}); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/gather.h b/paddle2onnx/mapper/tensor/gather.h deleted file mode 100644 index b12c7f5b2c7..00000000000 --- a/paddle2onnx/mapper/tensor/gather.h +++ /dev/null @@ -1,41 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class GatherMapper : public Mapper { - public: - GatherMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - if (HasAttr("axis")) { - GetAttr("axis", &axis_); - } - } - - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - void Opset11(); - - private: - int64_t axis_ = 0; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/gather_nd.cc b/paddle2onnx/mapper/tensor/gather_nd.cc deleted file mode 100755 index d7fcb875df2..00000000000 --- a/paddle2onnx/mapper/tensor/gather_nd.cc +++ /dev/null @@ -1,37 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/gather_nd.h" - -namespace paddle2onnx { -REGISTER_MAPPER(gather_nd, GatherNdMapper) - -int32_t GatherNdMapper::GetMinOpset(bool verbose) { - Logger(verbose, 11) << RequireOpset(11) << std::endl; - return 11; -} - -void GatherNdMapper::Opset11() { - auto input_x_info = GetInput("X"); - auto input_index_info = GetInput("Index"); - auto output_info = GetOutput("Out"); - - std::string index_node = helper_->AutoCast( - input_index_info[0].name, input_index_info[0].dtype, P2ODataType::INT64); - - helper_->MakeNode("GatherND", {input_x_info[0].name, index_node}, - {output_info[0].name}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/gather_nd.h b/paddle2onnx/mapper/tensor/gather_nd.h deleted file mode 100755 index ae7de4f5c4a..00000000000 --- a/paddle2onnx/mapper/tensor/gather_nd.h +++ /dev/null @@ -1,32 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class GatherNdMapper : public Mapper { - public: - GatherNdMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - int32_t GetMinOpset(bool verbose = false); - void Opset11(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/gaussian_random.cc b/paddle2onnx/mapper/tensor/gaussian_random.cc deleted file mode 100644 index 97ba7c50d64..00000000000 --- a/paddle2onnx/mapper/tensor/gaussian_random.cc +++ /dev/null @@ -1,80 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/gaussian_random.h" - -namespace paddle2onnx { -REGISTER_MAPPER(gaussian_random, GaussianRandomMapper) - -int32_t GaussianRandomMapper::GetMinOpset(bool verbose) { - if (HasInput("ShapeTensor") && !IsConstantInput("ShapeTensor")) { - Logger(verbose, 9) - << "While ShapeTensor as input and it's not a constant tensor, " - << RequireOpset(9) << std::endl; - return 9; - } - if (HasInput("ShapeTensorList")) { - Logger(verbose, 9) << "While ShapeTensorList as input, " << RequireOpset(9) - << std::endl; - return 9; - } - return 7; -} - -void GaussianRandomMapper::Opset7() { - auto out_info = GetOutput("Out"); - std::string shape_tensor_name = ""; - std::vector shape; - if (HasInput("ShapeTensor")) { - if (!TryGetInputValue("ShapeTensor", &shape)) { - auto shape_info = GetInput("ShapeTensor"); - shape_tensor_name = helper_->AutoCast( - shape_info[0].name, shape_info[0].dtype, P2ODataType::INT64); - } - } else if (HasInput("ShapeTensorList")) { - auto shape_info = GetInput("ShapeTensorList"); - shape_tensor_name = helper_->ConcatIndices(shape_info); - } else { - shape.assign(shape_.begin(), shape_.end()); - } - if (out_info[0].Rank() == 0) { - auto node = helper_->MakeNode("RandomNormal", {}); - AddAttribute(node, "dtype", GetOnnxDtype(out_info[0].dtype)); - AddAttribute(node, "mean", mean_); - AddAttribute(node, "scale", std_); - AddAttribute(node, "shape", std::vector(1, 1)); - AddAttribute(node, "seed", static_cast(seed_)); - helper_->Squeeze(node->output(0), {out_info[0].name}, {0}); - return; - } - if (shape.size() > 0) { - auto node = helper_->MakeNode("RandomNormal", {}, {out_info[0].name}); - AddAttribute(node, "dtype", GetOnnxDtype(out_info[0].dtype)); - AddAttribute(node, "mean", mean_); - AddAttribute(node, "scale", std_); - AddAttribute(node, "shape", shape_); - AddAttribute(node, "seed", static_cast(seed_)); - } else { - auto tensor = helper_->ConstOfShape( - shape_tensor_name, GetOnnxDtype(out_info[0].dtype), float(0)); - auto node = - helper_->MakeNode("RandomNormalLike", {tensor}, {out_info[0].name}); - AddAttribute(node, "dtype", GetOnnxDtype(out_info[0].dtype)); - AddAttribute(node, "mean", mean_); - AddAttribute(node, "scale", std_); - AddAttribute(node, "seed", static_cast(seed_)); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/gaussian_random.h b/paddle2onnx/mapper/tensor/gaussian_random.h deleted file mode 100644 index 0fde072ed05..00000000000 --- a/paddle2onnx/mapper/tensor/gaussian_random.h +++ /dev/null @@ -1,40 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class GaussianRandomMapper : public Mapper { - public: - GaussianRandomMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("mean", &mean_); - GetAttr("std", &std_); - GetAttr("shape", &shape_); - GetAttr("seed", &seed_); - } - - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - private: - std::vector shape_; - float mean_; - float std_; - int64_t seed_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/greater_equal.cc b/paddle2onnx/mapper/tensor/greater_equal.cc deleted file mode 100644 index ac386755b0a..00000000000 --- a/paddle2onnx/mapper/tensor/greater_equal.cc +++ /dev/null @@ -1,51 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/greater_equal.h" - -namespace paddle2onnx { -REGISTER_MAPPER(greater_equal, GreaterEqualMapper) - -void GreaterEqualMapper::Opset7() { - auto x_info = GetInput("X"); - auto y_info = GetInput("Y"); - auto out_info = GetOutput("Out"); - - int out_dtype = 0; - std::vector aligned_inputs = - helper_->DtypeAlignment({x_info[0], y_info[0]}, &out_dtype); - if (out_dtype != P2ODataType::FP32 && out_dtype != P2ODataType::FP64 && - helper_->GetOpsetVersion() < 11) { - aligned_inputs[0] = - helper_->AutoCast(aligned_inputs[0], out_dtype, P2ODataType::FP32); - aligned_inputs[1] = - helper_->AutoCast(aligned_inputs[1], out_dtype, P2ODataType::FP32); - } - - auto out = helper_->MakeNode("Less", aligned_inputs)->output(0); - helper_->MakeNode("Not", {out}, {out_info[0].name}); -} - -void GreaterEqualMapper::Opset12() { - auto x_info = GetInput("X"); - auto y_info = GetInput("Y"); - auto out_info = GetOutput("Out"); - - int out_dtype = 0; - std::vector aligned_inputs = - helper_->DtypeAlignment({x_info[0], y_info[0]}, &out_dtype); - - helper_->MakeNode("GreaterOrEqual", aligned_inputs, {out_info[0].name}); -} -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/greater_equal.h b/paddle2onnx/mapper/tensor/greater_equal.h deleted file mode 100644 index 200ea6bd8c6..00000000000 --- a/paddle2onnx/mapper/tensor/greater_equal.h +++ /dev/null @@ -1,29 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class GreaterEqualMapper : public Mapper { - public: - GreaterEqualMapper(const PaddleParser& p, OnnxHelper* helper, - int64_t block_id, int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); - void Opset12(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/greater_than.cc b/paddle2onnx/mapper/tensor/greater_than.cc deleted file mode 100644 index 36c4220a64f..00000000000 --- a/paddle2onnx/mapper/tensor/greater_than.cc +++ /dev/null @@ -1,39 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/greater_than.h" - -namespace paddle2onnx { -REGISTER_MAPPER(greater_than, GreaterThanMapper) - -void GreaterThanMapper::Opset7() { - auto x_info = GetInput("X"); - auto y_info = GetInput("Y"); - auto out_info = GetOutput("Out"); - - int out_dtype = 0; - std::vector aligned_inputs = - helper_->DtypeAlignment({x_info[0], y_info[0]}, &out_dtype); - if (out_dtype != P2ODataType::FP32 && out_dtype != P2ODataType::FP64 && - helper_->GetOpsetVersion() < 11) { - aligned_inputs[0] = - helper_->AutoCast(aligned_inputs[0], out_dtype, P2ODataType::FP32); - aligned_inputs[1] = - helper_->AutoCast(aligned_inputs[1], out_dtype, P2ODataType::FP32); - } - - helper_->MakeNode("Greater", aligned_inputs, {out_info[0].name}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/greater_than.h b/paddle2onnx/mapper/tensor/greater_than.h deleted file mode 100644 index c9867d864c4..00000000000 --- a/paddle2onnx/mapper/tensor/greater_than.h +++ /dev/null @@ -1,28 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class GreaterThanMapper : public Mapper { - public: - GreaterThanMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/grid_sampler.cc b/paddle2onnx/mapper/tensor/grid_sampler.cc deleted file mode 100644 index 0a08e4cdacf..00000000000 --- a/paddle2onnx/mapper/tensor/grid_sampler.cc +++ /dev/null @@ -1,41 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/grid_sampler.h" - -namespace paddle2onnx { -REGISTER_MAPPER(grid_sampler, GridSamplerMapper) - -int32_t GridSamplerMapper::GetMinOpset(bool verbose) { - Logger(verbose, 16) << RequireOpset(16) << std::endl; - return 16; -}; - -void GridSamplerMapper::Opset16() { - auto x_info = GetInput("X"); - auto grid_info = GetInput("Grid"); - auto out_info = GetOutput("Output"); - std::string cast_input = - helper_->AutoCast(x_info[0].name, x_info[0].dtype, P2ODataType::FP32); - std::string cast_grid = helper_->AutoCast( - grid_info[0].name, grid_info[0].dtype, P2ODataType::FP32); - auto node = helper_->MakeNode("GridSample", {cast_input, cast_grid}); - AddAttribute(node, "padding_mode", padding_mode_); - AddAttribute(node, "mode", mode_); - AddAttribute(node, "align_corners", static_cast(align_corners_)); - helper_->AutoCast(node->output(0), out_info[0].name, P2ODataType::FP32, - out_info[0].dtype); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/grid_sampler.h b/paddle2onnx/mapper/tensor/grid_sampler.h deleted file mode 100644 index 28da730ca3d..00000000000 --- a/paddle2onnx/mapper/tensor/grid_sampler.h +++ /dev/null @@ -1,43 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class GridSamplerMapper : public Mapper { - public: - GridSamplerMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("padding_mode", &padding_mode_); - GetAttr("mode", &mode_); - GetAttr("align_corners", &align_corners_); - } - - int32_t GetMinOpset(bool verbose = false); - - void Opset16(); - - private: - std::string padding_mode_ = "zeros"; - std::string mode_ = "bilinear"; - bool align_corners_ = false; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/index_sample.cc b/paddle2onnx/mapper/tensor/index_sample.cc deleted file mode 100644 index 1522dcbefe4..00000000000 --- a/paddle2onnx/mapper/tensor/index_sample.cc +++ /dev/null @@ -1,44 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/index_sample.h" - -namespace paddle2onnx { -REGISTER_MAPPER(index_sample, IndexSampleMapper) - -int32_t IndexSampleMapper::GetMinOpset(bool verbose) { - auto x_info = GetInput("X"); - auto index_info = GetInput("Index"); - if (x_info[0].Rank() != 2 || index_info[0].Rank() != 2) { - Error() << "The rank of X and Index must be 2, but the rank of X is: " - << x_info[0].Rank() - << " , and the rank of Index is: " << index_info[0].Rank() << "." - << std::endl; - return -1; - } - Logger(verbose, 11) << RequireOpset(11) << std::endl; - return 11; -} - -void IndexSampleMapper::Opset11() { - auto x_info = GetInput("X"); - auto index_info = GetInput("Index"); - auto out_info = GetOutput("Out"); - auto node = - helper_->MakeNode("GatherElements", {x_info[0].name, index_info[0].name}, - {out_info[0].name}); - AddAttribute(node, "axis", static_cast(1)); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/index_sample.h b/paddle2onnx/mapper/tensor/index_sample.h deleted file mode 100755 index f6bcaabd5a0..00000000000 --- a/paddle2onnx/mapper/tensor/index_sample.h +++ /dev/null @@ -1,32 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class IndexSampleMapper : public Mapper { - public: - IndexSampleMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - int32_t GetMinOpset(bool verbose = false); - void Opset11(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/index_select.cc b/paddle2onnx/mapper/tensor/index_select.cc deleted file mode 100644 index 3ff40cd2abb..00000000000 --- a/paddle2onnx/mapper/tensor/index_select.cc +++ /dev/null @@ -1,29 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/index_select.h" - -namespace paddle2onnx { -REGISTER_MAPPER(index_select, IndexSelectMapper) - -void IndexSelectMapper::Opset7() { - auto x_info = GetInput("X"); - auto index_info = GetInput("Index"); - auto out_info = GetOutput("Out"); - auto node = helper_->MakeNode("Gather", {x_info[0].name, index_info[0].name}, - {out_info[0].name}); - AddAttribute(node, "axis", axis_); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/index_select.h b/paddle2onnx/mapper/tensor/index_select.h deleted file mode 100644 index 463718a736d..00000000000 --- a/paddle2onnx/mapper/tensor/index_select.h +++ /dev/null @@ -1,36 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class IndexSelectMapper : public Mapper { - public: - IndexSelectMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("dim", &axis_); - } - void Opset7(); - - private: - int64_t axis_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/less_equal.cc b/paddle2onnx/mapper/tensor/less_equal.cc deleted file mode 100644 index 43abd8f8b59..00000000000 --- a/paddle2onnx/mapper/tensor/less_equal.cc +++ /dev/null @@ -1,51 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/less_equal.h" - -namespace paddle2onnx { -REGISTER_MAPPER(less_equal, LessEqualMapper) - -void LessEqualMapper::Opset7() { - auto x_info = GetInput("X"); - auto y_info = GetInput("Y"); - auto out_info = GetOutput("Out"); - - int out_dtype = 0; - std::vector aligned_inputs = - helper_->DtypeAlignment({x_info[0], y_info[0]}, &out_dtype); - if (out_dtype != P2ODataType::FP32 && out_dtype != P2ODataType::FP64 && - helper_->GetOpsetVersion() < 11) { - aligned_inputs[0] = - helper_->AutoCast(aligned_inputs[0], out_dtype, P2ODataType::FP32); - aligned_inputs[1] = - helper_->AutoCast(aligned_inputs[1], out_dtype, P2ODataType::FP32); - } - - auto out = helper_->MakeNode("Greater", aligned_inputs)->output(0); - helper_->MakeNode("Not", {out}, {out_info[0].name}); -} - -void LessEqualMapper::Opset12() { - auto x_info = GetInput("X"); - auto y_info = GetInput("Y"); - auto out_info = GetOutput("Out"); - - int out_dtype = 0; - std::vector aligned_inputs = - helper_->DtypeAlignment({x_info[0], y_info[0]}, &out_dtype); - - helper_->MakeNode("LessOrEqual", aligned_inputs, {out_info[0].name}); -} -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/less_equal.h b/paddle2onnx/mapper/tensor/less_equal.h deleted file mode 100644 index c1a26542567..00000000000 --- a/paddle2onnx/mapper/tensor/less_equal.h +++ /dev/null @@ -1,29 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class LessEqualMapper : public Mapper { - public: - LessEqualMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); - void Opset12(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/less_than.cc b/paddle2onnx/mapper/tensor/less_than.cc deleted file mode 100644 index f0db74dcac6..00000000000 --- a/paddle2onnx/mapper/tensor/less_than.cc +++ /dev/null @@ -1,39 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/less_than.h" - -namespace paddle2onnx { -REGISTER_MAPPER(less_than, LessThanMapper) - -void LessThanMapper::Opset7() { - auto x_info = GetInput("X"); - auto y_info = GetInput("Y"); - auto out_info = GetOutput("Out"); - - int out_dtype = 0; - std::vector aligned_inputs = - helper_->DtypeAlignment({x_info[0], y_info[0]}, &out_dtype); - if (out_dtype != P2ODataType::FP32 && out_dtype != P2ODataType::FP64 && - helper_->GetOpsetVersion() < 11) { - aligned_inputs[0] = - helper_->AutoCast(aligned_inputs[0], out_dtype, P2ODataType::FP32); - aligned_inputs[1] = - helper_->AutoCast(aligned_inputs[1], out_dtype, P2ODataType::FP32); - } - - helper_->MakeNode("Less", aligned_inputs, {out_info[0].name}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/less_than.h b/paddle2onnx/mapper/tensor/less_than.h deleted file mode 100644 index a25432d1b16..00000000000 --- a/paddle2onnx/mapper/tensor/less_than.h +++ /dev/null @@ -1,28 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class LessThanMapper : public Mapper { - public: - LessThanMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/linspace.cc b/paddle2onnx/mapper/tensor/linspace.cc deleted file mode 100644 index 9d62fb058f6..00000000000 --- a/paddle2onnx/mapper/tensor/linspace.cc +++ /dev/null @@ -1,75 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/linspace.h" - -#include -#include - -namespace paddle2onnx { -REGISTER_MAPPER(linspace, LinspaceMapper) - -int32_t LinspaceMapper::GetMinOpset(bool verbose) { - Logger(verbose, 9) << RequireOpset(9) << std::endl; - return 9; -}; - -void LinspaceMapper::Opset9() { - auto start_info = GetInput("Start"); - auto stop_info = GetInput("Stop"); - auto num_info = GetInput("Num"); - auto output_info = GetOutput("Out"); - - std::string cast_start = helper_->AutoCast( - start_info[0].name, start_info[0].dtype, P2ODataType::FP32); - std::string cast_stop = helper_->AutoCast( - stop_info[0].name, stop_info[0].dtype, P2ODataType::FP32); - - auto sub_a_node = helper_->MakeNode("Sub", {cast_stop, cast_start}); - - std::string one_node = helper_->Constant(GetOnnxDtype(num_info[0].dtype), - std::vector(1, 1)); - - auto sub_b_node = helper_->MakeNode("Sub", {num_info[0].name, one_node}); - - std::string sub_b_float_node = helper_->AutoCast( - sub_b_node->output(0), num_info[0].dtype, P2ODataType::FP32); - - auto step = - helper_->MakeNode("Div", {sub_a_node->output(0), sub_b_float_node}); - - std::string range_tensor = helper_->AutoCast( - num_info[0].name, num_info[0].dtype, P2ODataType::INT64); - - std::string one_like_node = helper_->ConstOfShape( - range_tensor, GetOnnxDtype(P2ODataType::FP32), static_cast(1)); - - auto none_zero_node = helper_->MakeNode("NonZero", {one_like_node}); - - std::string trans_squeeze = - helper_->Squeeze(none_zero_node->output(0), std::vector(1, 0)); - - std::string cast_trans_squeeze = - helper_->AutoCast(trans_squeeze, P2ODataType::INT64, P2ODataType::FP32); - - auto mul_node = - helper_->MakeNode("Mul", {cast_trans_squeeze, step->output(0)}); - - auto add_node = helper_->MakeNode("Add", {mul_node->output(0), cast_start}); - - helper_->AutoCast(add_node->output(0), output_info[0].name, P2ODataType::FP32, - output_info[0].dtype); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/linspace.h b/paddle2onnx/mapper/tensor/linspace.h deleted file mode 100644 index b43c97d472f..00000000000 --- a/paddle2onnx/mapper/tensor/linspace.h +++ /dev/null @@ -1,37 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class LinspaceMapper : public Mapper { - public: - LinspaceMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("dtype", &dtype_); - } - int32_t GetMinOpset(bool verbose = false); - void Opset9(); - - private: - int64_t dtype_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/logical_not.cc b/paddle2onnx/mapper/tensor/logical_not.cc deleted file mode 100644 index c1b79aba01b..00000000000 --- a/paddle2onnx/mapper/tensor/logical_not.cc +++ /dev/null @@ -1,27 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/logical_not.h" - -namespace paddle2onnx { -REGISTER_MAPPER(logical_not, LogicalNotMapper) - -void LogicalNotMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - helper_->MakeNode("Not", {input_info[0].name}, {output_info[0].name}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/logical_not.h b/paddle2onnx/mapper/tensor/logical_not.h deleted file mode 100644 index f667628fb5a..00000000000 --- a/paddle2onnx/mapper/tensor/logical_not.h +++ /dev/null @@ -1,31 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class LogicalNotMapper : public Mapper { - public: - LogicalNotMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/logical_op.cc b/paddle2onnx/mapper/tensor/logical_op.cc deleted file mode 100644 index 5f0646228a6..00000000000 --- a/paddle2onnx/mapper/tensor/logical_op.cc +++ /dev/null @@ -1,36 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/logical_op.h" - -namespace paddle2onnx { -REGISTER_MAPPER(logical_and, LogicalOpMapper) -REGISTER_MAPPER(logical_or, LogicalOpMapper) -REGISTER_MAPPER(logical_xor, LogicalOpMapper) - -void LogicalOpMapper::Opset7() { - auto x_info = GetInput("X"); - auto y_info = GetInput("Y"); - auto out_info = GetOutput("Out"); - - std::map op_mapper; - op_mapper["logical_and"] = "And"; - op_mapper["logical_or"] = "Or"; - op_mapper["logical_xor"] = "Xor"; - - helper_->MakeNode(op_mapper[OpType()], {x_info[0].name, y_info[0].name}, - {out_info[0].name}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/logical_op.h b/paddle2onnx/mapper/tensor/logical_op.h deleted file mode 100644 index 163b6685142..00000000000 --- a/paddle2onnx/mapper/tensor/logical_op.h +++ /dev/null @@ -1,31 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class LogicalOpMapper : public Mapper { - public: - LogicalOpMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/lookup_table.cc b/paddle2onnx/mapper/tensor/lookup_table.cc deleted file mode 100644 index 5342ba2ef1a..00000000000 --- a/paddle2onnx/mapper/tensor/lookup_table.cc +++ /dev/null @@ -1,132 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/lookup_table.h" - -#include -#include -#include - -namespace paddle2onnx { -REGISTER_MAPPER(lookup_table, LookupTableMapper) -REGISTER_MAPPER(lookup_table_v2, LookupTableMapper) - -int32_t LookupTableMapper::GetMinOpset(bool verbose) { - auto input_w_info = GetInput("W"); - bool has_minus = false; - for (auto i : input_w_info[0].shape) { - has_minus = (i == -1); - if (has_minus) { - break; - } - } - if (padding_idx_ != -1 && has_minus) { - Logger(verbose, 11) - << "While the input W has dynamic shape and padding_idx != -1, " - << RequireOpset(11) << std::endl; - return 11; - } - return 7; -} - -void LookupTableMapper::Opset7() { - auto input_ids_info = GetInput("Ids"); - auto input_w_info = GetInput("W"); - auto output_info = GetOutput("Out"); - - std::string ids_node = input_ids_info[0].name; - auto ids_shape = input_ids_info[0].shape; - if (OpType() == "lookup_table" && ids_shape[ids_shape.size() - 1] == 1) { - ids_node = helper_->Squeeze(input_ids_info[0].name, {-1}); - } - - auto input_shape = input_w_info[0].shape; - int64_t sum_val = 1; - for (auto i : input_shape) { - sum_val *= i; - } - int interval = sum_val / input_shape[0]; - - if (padding_idx_ != -1) { - std::vector data(sum_val, 1); - for (auto i = 0; i < interval; i++) { - data[padding_idx_ * interval + i] = 0; - } - std::string constant = helper_->Constant( - input_shape, GetOnnxDtype(input_w_info[0].dtype), data); - auto weight_node = - helper_->MakeNode("Mul", {input_w_info[0].name, constant}); - helper_->MakeNode("Gather", {weight_node->output(0), ids_node}, - {output_info[0].name}); - } else { - helper_->MakeNode("Gather", {input_w_info[0].name, ids_node}, - {output_info[0].name}); - } -} - -void LookupTableMapper::Opset11() { - auto input_ids_info = GetInput("Ids"); - auto input_w_info = GetInput("W"); - auto output_info = GetOutput("Out"); - - std::string ids_node = input_ids_info[0].name; - auto ids_shape = input_ids_info[0].shape; - if (OpType() == "lookup_table" && ids_shape[ids_shape.size() - 1] == 1) { - ids_node = helper_->Squeeze(input_ids_info[0].name, {-1}); - } - - auto input_shape = input_w_info[0].shape; - int64_t sum_val = 1; - for (auto i : input_shape) { - sum_val *= i; - } - int interval = sum_val / input_shape[0]; - - if (padding_idx_ != -1) { - bool has_minus = false; - for (auto i : input_w_info[0].shape) { - has_minus = (i == -1); - if (has_minus) { - break; - } - } - if (has_minus) { - std::vector shape = {interval}; - std::string replace_data = - helper_->Constant(shape, GetOnnxDtype(input_w_info[0].dtype), 0.0); - std::string index = helper_->Constant( - {1}, ONNX_NAMESPACE::TensorProto::INT64, padding_idx_); - auto scatter_node = helper_->MakeNode( - "ScatterND", {input_w_info[0].name, index, replace_data}); - helper_->MakeNode("Gather", {scatter_node->output(0), ids_node}, - {output_info[0].name}); - } else { - std::vector data(sum_val, 1); - for (auto i = 0; i < interval; i++) { - data[padding_idx_ * interval + i] = 0; - } - std::string constant = helper_->Constant( - input_shape, GetOnnxDtype(input_w_info[0].dtype), data); - auto weight_node = - helper_->MakeNode("Mul", {input_w_info[0].name, constant}); - helper_->MakeNode("Gather", {weight_node->output(0), ids_node}, - {output_info[0].name}); - } - } else { - helper_->MakeNode("Gather", {input_w_info[0].name, ids_node}, - {output_info[0].name}); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/lookup_table.h b/paddle2onnx/mapper/tensor/lookup_table.h deleted file mode 100644 index df537a981ee..00000000000 --- a/paddle2onnx/mapper/tensor/lookup_table.h +++ /dev/null @@ -1,39 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class LookupTableMapper : public Mapper { - public: - LookupTableMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("padding_idx", &padding_idx_); - } - - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - void Opset11(); - - private: - int64_t padding_idx_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/matmul.cc b/paddle2onnx/mapper/tensor/matmul.cc deleted file mode 100644 index fd8bb126444..00000000000 --- a/paddle2onnx/mapper/tensor/matmul.cc +++ /dev/null @@ -1,61 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/matmul.h" -#include - -namespace paddle2onnx { -REGISTER_MAPPER(matmul, MatmulMapper) - -std::string MatmulMapper::GetTrans(std::vector& input_info) { - std::string castd_name = input_info[0].name; - if (input_info[0].dtype == P2ODataType::FP64) { - castd_name = helper_->AutoCast(input_info[0].name, input_info[0].dtype, - P2ODataType::FP32); - } - std::vector perm = Arange(0, input_info[0].Rank()); - std::swap(perm[perm.size() - 1], perm[perm.size() - 2]); - auto transpose_node = helper_->MakeNode("Transpose", {castd_name}); - AddAttribute(transpose_node, "perm", perm); - return transpose_node->output(0); -} - -void MatmulMapper::Opset7() { - auto input_x_info = GetInput("X"); - auto input_y_info = GetInput("Y"); - auto output_info = GetOutput("Out"); - std::string input_x = input_x_info[0].name; - if (transpose_X_) { - input_x = GetTrans(input_x_info); - } - std::string input_y = input_y_info[0].name; - if (transpose_Y_) { - input_y = GetTrans(input_y_info); - } - if (fabs(alpha_ - 1.0) < 1e-6) { - auto node = helper_->MakeNode("MatMul", {input_x, input_y}); - helper_->AutoCast(node->output(0), output_info[0].name, P2ODataType::FP32, - input_y_info[0].dtype); - } else { - auto mutmul_node = helper_->MakeNode("MatMul", {input_x, input_y}); - std::string scale_node = - helper_->Constant({1}, GetOnnxDtype(input_x_info[0].dtype), alpha_); - auto mul_node = - helper_->MakeNode("Mul", {mutmul_node->output(0), scale_node}); - helper_->AutoCast(mul_node->output(0), output_info[0].name, - P2ODataType::FP32, input_y_info[0].dtype); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/matmul.h b/paddle2onnx/mapper/tensor/matmul.h deleted file mode 100644 index 16957701fbc..00000000000 --- a/paddle2onnx/mapper/tensor/matmul.h +++ /dev/null @@ -1,42 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class MatmulMapper : public Mapper { - public: - MatmulMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("transpose_X", &transpose_X_); - GetAttr("transpose_Y", &transpose_Y_); - GetAttr("alpha", &alpha_); - } - - void Opset7(); - - private: - std::string GetTrans(std::vector& input_info); - bool transpose_X_ = false; - bool transpose_Y_ = false; - float alpha_ = 1.0; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/matmul_v2.cc b/paddle2onnx/mapper/tensor/matmul_v2.cc deleted file mode 100644 index f78a26f4241..00000000000 --- a/paddle2onnx/mapper/tensor/matmul_v2.cc +++ /dev/null @@ -1,51 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/matmul_v2.h" - -#include -#include -#include - -namespace paddle2onnx { -REGISTER_MAPPER(matmul_v2, MatmulV2Mapper) - -std::string MatmulV2Mapper::GetTrans(std::vector& input_info) { - std::string castd_name = helper_->AutoCast( - input_info[0].name, input_info[0].dtype, P2ODataType::FP32); - std::vector perm = Arange(0, input_info[0].Rank()); - std::swap(perm[perm.size() - 1], perm[perm.size() - 2]); - auto transpose_node = helper_->MakeNode("Transpose", {castd_name}); - AddAttribute(transpose_node, "perm", perm); - return transpose_node->output(0); -} - -void MatmulV2Mapper::Opset7() { - auto input_x_info = GetInput("X"); - auto input_y_info = GetInput("Y"); - auto output_info = GetOutput("Out"); - std::string input_x = input_x_info[0].name; - if (trans_x_) { - input_x = GetTrans(input_x_info); - } - std::string input_y = input_y_info[0].name; - if (trans_y_) { - input_y = GetTrans(input_y_info); - } - auto node = helper_->MakeNode("MatMul", {input_x, input_y}); - helper_->AutoCast(node->output(0), output_info[0].name, P2ODataType::FP32, - input_y_info[0].dtype); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/matmul_v2.h b/paddle2onnx/mapper/tensor/matmul_v2.h deleted file mode 100644 index bb3762a34dc..00000000000 --- a/paddle2onnx/mapper/tensor/matmul_v2.h +++ /dev/null @@ -1,40 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class MatmulV2Mapper : public Mapper { - public: - MatmulV2Mapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("trans_x", &trans_x_); - GetAttr("trans_y", &trans_y_); - } - - void Opset7(); - - private: - std::string GetTrans(std::vector& input_info); - bool trans_x_ = false; - bool trans_y_ = false; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/mean.cc b/paddle2onnx/mapper/tensor/mean.cc deleted file mode 100644 index de1fae3cbe7..00000000000 --- a/paddle2onnx/mapper/tensor/mean.cc +++ /dev/null @@ -1,29 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/mean.h" - -namespace paddle2onnx { -REGISTER_MAPPER(mean, MeanMapper) - -void MeanMapper::Opset7() { - auto x_info = GetInput("X"); - auto out_info = GetOutput("Out"); - auto input = helper_->Reshape(x_info[0].name, {-1}); - auto node = helper_->MakeNode("ReduceMean", {input}, {out_info[0].name}); - AddAttribute(node, "axes", std::vector(1, 0)); - AddAttribute(node, "keepdims", int64_t(1)); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/mean.h b/paddle2onnx/mapper/tensor/mean.h deleted file mode 100644 index bbb6a2bd8d4..00000000000 --- a/paddle2onnx/mapper/tensor/mean.h +++ /dev/null @@ -1,31 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class MeanMapper : public Mapper { - public: - MeanMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/meshgrid.cc b/paddle2onnx/mapper/tensor/meshgrid.cc deleted file mode 100644 index b855d1544b6..00000000000 --- a/paddle2onnx/mapper/tensor/meshgrid.cc +++ /dev/null @@ -1,46 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/meshgrid.h" - -namespace paddle2onnx { -REGISTER_MAPPER(meshgrid, MeshgridMapper) - -void MeshgridMapper::Opset8() { - auto x_info = GetInput("X"); - auto out_info = GetOutput("Out"); - - std::vector x_shapes(x_info.size()); - for (size_t i = 0; i < x_info.size(); ++i) { - x_shapes[i] = helper_->MakeNode("Shape", {x_info[i].name})->output(0); - } - auto out_shape = helper_->Concat(x_shapes, 0); - for (size_t i = 0; i < x_info.size(); ++i) { - std::vector intermediate_shape(x_info.size()); - for (size_t j = 0; j < x_info.size(); ++j) { - if (j == i) { - intermediate_shape[j] = x_shapes[i]; - } else { - intermediate_shape[j] = helper_->Constant( - ONNX_NAMESPACE::TensorProto::INT64, std::vector(1, 1)); - } - } - auto t_reshaped = helper_->Concat(intermediate_shape, 0); - t_reshaped = - helper_->MakeNode("Reshape", {x_info[i].name, t_reshaped})->output(0); - helper_->MakeNode("Expand", {t_reshaped, out_shape}, {out_info[i].name}); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/meshgrid.h b/paddle2onnx/mapper/tensor/meshgrid.h deleted file mode 100644 index d83d677b00c..00000000000 --- a/paddle2onnx/mapper/tensor/meshgrid.h +++ /dev/null @@ -1,35 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class MeshgridMapper : public Mapper { - public: - MeshgridMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - MarkAsExperimentalOp(); - } - - int32_t GetMinOpset(bool verbose = false) { return 8; } - void Opset8(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/mul.cc b/paddle2onnx/mapper/tensor/mul.cc deleted file mode 100644 index 3ba9e1212ee..00000000000 --- a/paddle2onnx/mapper/tensor/mul.cc +++ /dev/null @@ -1,50 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/mul.h" - -namespace paddle2onnx { -REGISTER_MAPPER(mul, MulMapper) - -void MulMapper::Opset7() { - auto x_info = GetInput("X"); - auto y_info = GetInput("Y"); - auto out_info = GetOutput("Out"); - - auto x_input = x_info[0].name; - auto y_input = y_info[0].name; - if (x_info[0].Rank() > 2) { - auto node = helper_->MakeNode("Flatten", {x_input}); - AddAttribute(node, "axis", x_num_col_dims_); - x_input = node->output(0); - } - if (y_info[0].Rank() > 2) { - auto node = helper_->MakeNode("Flatten", {y_input}); - AddAttribute(node, "axis", y_num_col_dims_); - y_input = node->output(0); - } - auto out = helper_->MakeNode("MatMul", {x_input, y_input})->output(0); - - if (x_info[0].Rank() != 2 || y_info[0].Rank() != 2) { - auto x_shape = helper_->MakeNode("Shape", {x_info[0].name})->output(0); - auto y_shape = helper_->MakeNode("Shape", {y_info[0].name})->output(0); - auto out_shape_0 = helper_->Slice(x_shape, {0}, {0}, {x_num_col_dims_}); - auto out_shape_1 = helper_->Slice(y_shape, {0}, {y_num_col_dims_}, {y_info[0].Rank()}); - auto out_shape = helper_->Concat({out_shape_0, out_shape_1}, 0); - out = helper_->MakeNode("Reshape", {out, out_shape})->output(0); - } - helper_->MakeNode("Identity", {out}, {out_info[0].name}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/mul.h b/paddle2onnx/mapper/tensor/mul.h deleted file mode 100644 index ad3171948a0..00000000000 --- a/paddle2onnx/mapper/tensor/mul.h +++ /dev/null @@ -1,37 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class MulMapper : public Mapper { - public: - MulMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("x_num_col_dims", &x_num_col_dims_); - GetAttr("y_num_col_dims", &y_num_col_dims_); - } - void Opset7(); - private: - int64_t x_num_col_dims_ = 1; - int64_t y_num_col_dims_ = 1; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/mv.cc b/paddle2onnx/mapper/tensor/mv.cc deleted file mode 100644 index dcbbdb8e306..00000000000 --- a/paddle2onnx/mapper/tensor/mv.cc +++ /dev/null @@ -1,33 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/mv.h" - -#include -#include -#include - -namespace paddle2onnx { -REGISTER_MAPPER(mv, MVMapper) - -void MVMapper::Opset7() { - auto input_x_info = GetInput("X"); - auto input_y_info = GetInput("Vec"); - auto output_info = GetOutput("Out"); - auto node = - helper_->MakeNode("MatMul", {input_x_info[0].name, input_y_info[0].name}, - {output_info[0].name}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/mv.h b/paddle2onnx/mapper/tensor/mv.h deleted file mode 100644 index 20acbf3c3b9..00000000000 --- a/paddle2onnx/mapper/tensor/mv.h +++ /dev/null @@ -1,32 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class MVMapper : public Mapper { - public: - MVMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - - void Opset7(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/nonzero.cc b/paddle2onnx/mapper/tensor/nonzero.cc deleted file mode 100644 index fd4c0abed1f..00000000000 --- a/paddle2onnx/mapper/tensor/nonzero.cc +++ /dev/null @@ -1,28 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/nonzero.h" - -namespace paddle2onnx { -REGISTER_MAPPER(where_index, NonZeroMapper) - -void NonZeroMapper::Opset9() { - auto input_info = GetInput("Condition"); - auto output_info = GetOutput("Out"); - auto non_zero_indices = - helper_->MakeNode("NonZero", {input_info[0].name})->output(0); - helper_->Transpose(non_zero_indices, output_info[0].name, {1, 0}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/nonzero.h b/paddle2onnx/mapper/tensor/nonzero.h deleted file mode 100644 index a5c26dba245..00000000000 --- a/paddle2onnx/mapper/tensor/nonzero.h +++ /dev/null @@ -1,33 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class NonZeroMapper : public Mapper { - public: - NonZeroMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - int32_t GetMinOpset(bool verbose = false) { - Logger(verbose, 9) << RequireOpset(9) << std::endl; - return 9; - } - void Opset9(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/not_equal.cc b/paddle2onnx/mapper/tensor/not_equal.cc deleted file mode 100644 index 66b987f1b94..00000000000 --- a/paddle2onnx/mapper/tensor/not_equal.cc +++ /dev/null @@ -1,45 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/not_equal.h" - -namespace paddle2onnx { -REGISTER_MAPPER(not_equal, NotEqualMapper) - -int32_t NotEqualMapper::GetMinOpset(bool verbose) { - auto x_info = GetInput("X"); - auto y_info = GetInput("Y"); - if (x_info[0].dtype == P2ODataType::FP32 || - x_info[0].dtype == P2ODataType::FP64) { - Logger(verbose, 11) << "While input is dtype of float32/float64, " - << RequireOpset(11) << std::endl; - return 11; - } - return 7; -} - -void NotEqualMapper::Opset7() { - auto x_info = GetInput("X"); - auto y_info = GetInput("Y"); - auto out_info = GetOutput("Out"); - - int out_dtype = 0; - std::vector aligned_inputs = - helper_->DtypeAlignment({x_info[0], y_info[0]}, &out_dtype); - - auto output = helper_->MakeNode("Equal", aligned_inputs)->output(0); - helper_->MakeNode("Not", {output}, {out_info[0].name}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/not_equal.h b/paddle2onnx/mapper/tensor/not_equal.h deleted file mode 100644 index 7c9f9c5f3d1..00000000000 --- a/paddle2onnx/mapper/tensor/not_equal.h +++ /dev/null @@ -1,29 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class NotEqualMapper : public Mapper { - public: - NotEqualMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - int32_t GetMinOpset(bool verbose = false); - void Opset7(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/one_hot_v2.cc b/paddle2onnx/mapper/tensor/one_hot_v2.cc deleted file mode 100644 index 6b877bb9b2d..00000000000 --- a/paddle2onnx/mapper/tensor/one_hot_v2.cc +++ /dev/null @@ -1,58 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/one_hot_v2.h" - -namespace paddle2onnx { -REGISTER_MAPPER(one_hot_v2, OneHotV2Mapper) - -int32_t OneHotV2Mapper::GetMinOpset(bool verbose) { - if (allow_out_of_range_) { - Error() << "allow_out_of_range is not supported in one_hot_v2." - << std::endl; - return -1; - } - auto output_info = GetOutput("Out"); - if (output_info[0].dtype != dtype_) { - Error() << "dtype attribute and output dtype do not match." << std::endl; - return -1; - } - Logger(verbose, 9) << RequireOpset(9) << std::endl; - return 9; -} - -void OneHotV2Mapper::Opset9() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - std::string casted_input = helper_->AutoCast( - input_info[0].name, input_info[0].dtype, P2ODataType::INT64); - - std::vector vals = {0, 1}; - std::string value_node = - helper_->Constant(GetOnnxDtype(output_info[0].dtype), vals); - - std::string depth_node = ""; - if (HasInput("depth_tensor")) { - auto input_depth_info = GetInput("depth_tensor"); - depth_node = input_depth_info[0].name; - } else { - depth_node = - helper_->Constant({1}, GetOnnxDtype(input_info[0].dtype), depth_); - } - auto one_hot_node = helper_->MakeNode( - "OneHot", {casted_input, depth_node, value_node}, {output_info[0].name}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/one_hot_v2.h b/paddle2onnx/mapper/tensor/one_hot_v2.h deleted file mode 100644 index bf560210518..00000000000 --- a/paddle2onnx/mapper/tensor/one_hot_v2.h +++ /dev/null @@ -1,41 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class OneHotV2Mapper : public Mapper { - public: - OneHotV2Mapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("allow_out_of_range", &allow_out_of_range_); - GetAttr("depth", &depth_); - GetAttr("dtype", &dtype_); - } - int32_t GetMinOpset(bool verbose); - void Opset9(); - - private: - bool allow_out_of_range_; - int64_t depth_; - int64_t dtype_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/p_norm.cc b/paddle2onnx/mapper/tensor/p_norm.cc deleted file mode 100755 index 84bc84476f6..00000000000 --- a/paddle2onnx/mapper/tensor/p_norm.cc +++ /dev/null @@ -1,50 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/p_norm.h" - -namespace paddle2onnx { -REGISTER_MAPPER(p_norm, PNormMapper) - -void PNormMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - std::string pnode = - helper_->Constant({1}, GetOnnxDtype(input_info[0].dtype), porder_); - - auto abs_node = helper_->MakeNode("Abs", {input_info[0].name}); - auto pow_node = helper_->MakeNode("Pow", {abs_node->output(0), pnode}); - std::string reducesum_node = ""; - std::vector axes_val = {axis_}; - if (helper_->GetOpsetVersion() < 13) { - auto node = helper_->MakeNode("ReduceSum", {pow_node->output(0)}); - AddAttribute(node, "axes", axes_val); - AddAttribute(node, "keepdims", static_cast(keepdim_)); - reducesum_node = node->output(0); - } else { - std::string axes_node = - helper_->Constant(GetOnnxDtype(P2ODataType::INT64), axes_val); - auto node = - helper_->MakeNode("ReduceSum", {pow_node->output(0), axes_node}); - AddAttribute(node, "keepdims", static_cast(keepdim_)); - reducesum_node = node->output(0); - } - - std::string pnode1 = - helper_->Constant({1}, GetOnnxDtype(input_info[0].dtype), 1.0 / porder_); - helper_->MakeNode("Pow", {reducesum_node, pnode1}, {output_info[0].name}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/p_norm.h b/paddle2onnx/mapper/tensor/p_norm.h deleted file mode 100755 index e7417369c38..00000000000 --- a/paddle2onnx/mapper/tensor/p_norm.h +++ /dev/null @@ -1,40 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class PNormMapper : public Mapper { - public: - PNormMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("keepdim", &keepdim_); - GetAttr("axis", &axis_); - GetAttr("porder", &porder_); - } - void Opset7(); - - private: - bool keepdim_; - int64_t axis_; - float porder_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/partial_ops.cc b/paddle2onnx/mapper/tensor/partial_ops.cc deleted file mode 100755 index d22f3e73db4..00000000000 --- a/paddle2onnx/mapper/tensor/partial_ops.cc +++ /dev/null @@ -1,88 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/partial_ops.h" - -namespace paddle2onnx { -REGISTER_MAPPER(partial_sum, PartialOpsMapper) -REGISTER_MAPPER(partial_concat, PartialOpsMapper) - -int32_t PartialOpsMapper::GetMinOpset(bool verbose) { - auto input_info = GetInput("X"); - for (auto &in : input_info) { - if (in.Rank() != 2) { - Error() << "The input dim of partial_sum OP must be 2." << std::endl; - return -1; - } - } - if (start_index_ < 0) { - start_index_ = start_index_ + input_info[0].shape[1]; - } - int64_t batch_size = input_info[0].shape[0]; - int64_t max_length = input_info[0].shape[1]; - for (auto &in : input_info) { - if (in.shape[0] != batch_size || in.shape[1] != max_length) { - Error() - << "The batch_size and max_length of all inputs must be same in " + - OpType() + " OP." - << std::endl; - return -1; - } - } - if (max_length < start_index_) { - Error() << "start_index must be less than input len in " + OpType() + " OP." - << std::endl; - return -1; - } - if (length_ > 0 && start_index_ + length_ > max_length) { - Error() << "start_index + length is larger than input length in " + - OpType() + " OP." - << std::endl; - return -1; - } - auto iter = op_mapper_.find(OpType()); - if (op_mapper_.end() == iter) { - Error() << "Cannot find " + OpType() + " in partial op_mapper." - << std::endl; - return -1; - } - return 7; -} - -void PartialOpsMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - int64_t end; - if (length_ < 0) { - end = input_info[0].shape[1]; - } else { - end = start_index_ + length_; - } - std::vector slice_outputs; - for (auto &in : input_info) { - auto out = helper_->Slice(in.name, {1}, {start_index_}, {end}); - std::string casted_node = - helper_->AutoCast(out, in.dtype, P2ODataType::FP32); - slice_outputs.push_back(casted_node); - } - auto iter = op_mapper_.find(OpType()); - auto node = helper_->MakeNode(iter->second, slice_outputs); - if (iter->second == "Concat") { - AddAttribute(node, "axis", static_cast(1)); - } - helper_->AutoCast(node->output(0), {output_info[0].name}, P2ODataType::FP32, - output_info[0].dtype); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/partial_ops.h b/paddle2onnx/mapper/tensor/partial_ops.h deleted file mode 100755 index 54eb7ecc3a0..00000000000 --- a/paddle2onnx/mapper/tensor/partial_ops.h +++ /dev/null @@ -1,42 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class PartialOpsMapper : public Mapper { - public: - PartialOpsMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("start_index", &start_index_); - GetAttr("length", &length_); - op_mapper_["partial_sum"] = "Sum"; - op_mapper_["partial_concat"] = "Concat"; - } - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - std::map op_mapper_; - int64_t start_index_; - int64_t length_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/pixel_shuffle.cc b/paddle2onnx/mapper/tensor/pixel_shuffle.cc deleted file mode 100644 index 9f10087d115..00000000000 --- a/paddle2onnx/mapper/tensor/pixel_shuffle.cc +++ /dev/null @@ -1,30 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/pixel_shuffle.h" - -namespace paddle2onnx { -REGISTER_MAPPER(pixel_shuffle, PixelShuffleMapper) - -void PixelShuffleMapper::Opset11() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - auto node = helper_->MakeNode("DepthToSpace", {input_info[0].name}, - {output_info[0].name}); - AddAttribute(node, "blocksize", upscale_factor_); - AddAttribute(node, "mode", static_cast("CRD")); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/pixel_shuffle.h b/paddle2onnx/mapper/tensor/pixel_shuffle.h deleted file mode 100644 index 9158810604e..00000000000 --- a/paddle2onnx/mapper/tensor/pixel_shuffle.h +++ /dev/null @@ -1,41 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class PixelShuffleMapper : public Mapper { - public: - PixelShuffleMapper(const PaddleParser& p, OnnxHelper* helper, - int64_t block_id, int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("upscale_factor", &upscale_factor_); - } - - int32_t GetMinOpset(bool verbose = false) { - Logger(verbose, 11) << RequireOpset(11) << std::endl; - return 11; - } - void Opset11(); - - private: - int64_t upscale_factor_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/pow.cc b/paddle2onnx/mapper/tensor/pow.cc deleted file mode 100644 index 429b7e8d75f..00000000000 --- a/paddle2onnx/mapper/tensor/pow.cc +++ /dev/null @@ -1,40 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/pow.h" - -#include - -namespace paddle2onnx { -REGISTER_MAPPER(pow, PowMapper) - -void PowMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - auto factor_node = - helper_->Constant({}, ONNX_NAMESPACE::TensorProto::FLOAT, factor_); - if (input_info[0].dtype != P2ODataType::FP32) { - std::string x_cast_name = helper_->AutoCast( - {input_info[0].name}, input_info[0].dtype, P2ODataType::FP32); - auto node = helper_->MakeNode("Pow", {x_cast_name, factor_node}); - helper_->AutoCast(node->output(0), {output_info[0].name}, P2ODataType::FP32, - input_info[0].dtype); - } else { - helper_->MakeNode("Pow", {input_info[0].name, factor_node}, - {output_info[0].name}); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/pow.h b/paddle2onnx/mapper/tensor/pow.h deleted file mode 100644 index 60b44509468..00000000000 --- a/paddle2onnx/mapper/tensor/pow.h +++ /dev/null @@ -1,36 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class PowMapper : public Mapper { - public: - PowMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("factor", &factor_); - } - void Opset7(); - - private: - float factor_ = 0.0; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/range.cc b/paddle2onnx/mapper/tensor/range.cc deleted file mode 100644 index 0254bf6a05a..00000000000 --- a/paddle2onnx/mapper/tensor/range.cc +++ /dev/null @@ -1,76 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/range.h" - -namespace paddle2onnx { -REGISTER_MAPPER(range, RangeMapper) - -void RangeMapper::Opset11() { - auto start_info = GetInput("Start"); - auto end_info = GetInput("End"); - auto step_info = GetInput("Step"); - auto out_info = GetOutput("Out"); - int32_t out_dtype = -1; - // TODO(jiangjiajun) cast for constant is an eleminable operation - std::vector aligned_inputs = helper_->DtypeAlignment( - {start_info[0], end_info[0], step_info[0]}, &out_dtype); - std::vector empty_axes; - - // // Trick for tensorrt - // if (out_dtype == P2ODataType::INT32 || out_dtype == P2ODataType::INT64 || - // true) { - // if (start_info[0].Rank() != 1) { - // aligned_inputs[0] = helper_->Reshape(aligned_inputs[0], {-1}); - // } - // if (end_info[0].Rank() != 1) { - // aligned_inputs[1] = helper_->Reshape(aligned_inputs[1], {-1}); - // } - // if (step_info[0].Rank() != 1) { - // aligned_inputs[2] = helper_->Reshape(aligned_inputs[2], {-1}); - // } - // auto length = helper_->MakeNode("Sub", {aligned_inputs[1], - // aligned_inputs[0]})->output(0); - // length = helper_->AutoCast(length, out_dtype, P2ODataType::INT64); - // auto one = helper_->Constant({1}, GetOnnxDtype(out_dtype), int64_t(1)); - // auto expaned_one = helper_->MakeNode("Expand", {one, - // length})->output(0); auto axis = helper_->Constant({}, - // ONNX_NAMESPACE::TensorProto::INT64, int64_t(0)); auto cumsumed_data = - // helper_->MakeNode("CumSum", {expaned_one, axis})->output(0); - // cumsumed_data = helper_->MakeNode("Sub", {cumsumed_data, - // one})->output(0); - // - // auto zero = helper_->Constant({1}, ONNX_NAMESPACE::TensorProto::INT64, - // int64_t(0)); - // auto new_step = helper_->AutoCast(aligned_inputs[2], step_info[0].dtype, - // P2ODataType::INT64); - // helper_->MakeNode("Slice", {cumsumed_data, zero, length, zero, - // new_step}, {out_info[0].name}); return; - // } - - // TODO(jiangjiajun) squeeze for constant is an eleminable operation - if (start_info[0].shape.size() > 0) { - aligned_inputs[0] = helper_->Squeeze(aligned_inputs[0], empty_axes); - } - if (end_info[0].shape.size() > 0) { - aligned_inputs[1] = helper_->Squeeze(aligned_inputs[1], empty_axes); - } - if (step_info[0].shape.size() > 0) { - aligned_inputs[2] = helper_->Squeeze(aligned_inputs[2], empty_axes); - } - auto out = helper_->MakeNode("Range", aligned_inputs)->output(0); - helper_->AutoCast(out, out_info[0].name, out_dtype, out_info[0].dtype); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/range.h b/paddle2onnx/mapper/tensor/range.h deleted file mode 100644 index df5c2dfe329..00000000000 --- a/paddle2onnx/mapper/tensor/range.h +++ /dev/null @@ -1,36 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class RangeMapper : public Mapper { - public: - RangeMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - - int32_t GetMinOpset(bool verbose = false) { - Logger(verbose, 11) << RequireOpset(11) << std::endl; - return 11; - } - void Opset11(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/reduce.cc b/paddle2onnx/mapper/tensor/reduce.cc deleted file mode 100644 index 82b98619c1c..00000000000 --- a/paddle2onnx/mapper/tensor/reduce.cc +++ /dev/null @@ -1,144 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/reduce.h" - -namespace paddle2onnx { -REGISTER_MAPPER(reduce_mean, ReduceMapper) -REGISTER_MAPPER(reduce_sum, ReduceMapper) -REGISTER_MAPPER(reduce_min, ReduceMapper) -REGISTER_MAPPER(reduce_max, ReduceMapper) -REGISTER_MAPPER(reduce_prod, ReduceMapper) -REGISTER_MAPPER(logsumexp, ReduceMapper) -REGISTER_MAPPER(reduce_all, ReduceMapper) -REGISTER_MAPPER(reduce_any, ReduceMapper) - -int32_t ReduceMapper::GetMinOpset(bool verbose) { - std::string axis_name; - if (OpType() == "logsumexp") { - axis_name = "axis"; - } else { - axis_name = "dim"; - } - if (IsAttrVar(axis_name) && !IsConstant(GetAttrVar(axis_name)[0])) { - if (OpType() == "reduce_sum") { - return 13; - } - Error() << "While Attribute(" << axis_name - << ")'s type is Tensor, it's not supported " - "unless it's a constant tensor." - << std::endl; - return -1; - } - return 7; -} - -void ReduceMapper::Opset7() { - auto x_info = GetInput("X"); - auto out_info = GetOutput("Out"); - std::map op_map; - op_map["reduce_mean"] = "ReduceMean"; - op_map["reduce_sum"] = "ReduceSum"; - op_map["reduce_min"] = "ReduceMin"; - op_map["reduce_max"] = "ReduceMax"; - op_map["reduce_prod"] = "ReduceProd"; - op_map["logsumexp"] = "ReduceLogSumExp"; - std::string out = ""; - - std::string axis_name; - if (OpType() == "logsumexp") { - axis_name = "axis"; - } else { - axis_name = "dim"; - } - - if (IsAttrVar(axis_name)) { - auto info = GetAttrVar(axis_name); - TryGetValue(info[0], &dim_); - } else { - GetAttr(axis_name, &dim_); - } - - bool reduce_all_axes = dim_.size() == x_info[0].Rank(); - if (reduce_all_) { - reduce_all_axes = true; - } - - if (helper_->GetOpsetVersion() >= 13 && OpType() == "reduce_sum") { - std::string dims = ""; - if (IsAttrVar(axis_name)) { - auto info = GetAttrVar(axis_name); - dims = helper_->AutoCast(info[0].name, info[0].dtype, P2ODataType::INT64); - } else { - if (!reduce_all_) { - dims = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, dim_); - } else { - dims = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, - Arange(0, x_info[0].Rank())); - } - } - auto reduce_node = - helper_->MakeNode(op_map[OpType()], {x_info[0].name, dims}); - AddAttribute(reduce_node, "keepdims", static_cast(keep_dim_)); - out = reduce_node->output(0); - } else if (OpType() == "reduce_all") { - auto int32_x = - helper_->AutoCast(x_info[0].name, x_info[0].dtype, P2ODataType::INT32); - auto reduce_node = helper_->MakeNode("ReduceMin", {int32_x}); - if (!reduce_all_) { - AddAttribute(reduce_node, "axes", dim_); - } else { - AddAttribute(reduce_node, "axes", Arange(0, x_info[0].Rank())); - } - AddAttribute(reduce_node, "keepdims", static_cast(keep_dim_)); - out = helper_->AutoCast(reduce_node->output(0), P2ODataType::INT32, - P2ODataType::BOOL); - } else if (OpType() == "reduce_any") { - auto int32_x = - helper_->AutoCast(x_info[0].name, x_info[0].dtype, P2ODataType::INT32); - auto reduce_node = helper_->MakeNode("ReduceMax", {int32_x}); - if (!reduce_all_) { - AddAttribute(reduce_node, "axes", dim_); - } else { - AddAttribute(reduce_node, "axes", Arange(0, x_info[0].Rank())); - } - AddAttribute(reduce_node, "keepdims", static_cast(keep_dim_)); - out = helper_->AutoCast(reduce_node->output(0), P2ODataType::INT32, - P2ODataType::BOOL); - } else { - std::string input_name = x_info[0].name; - if (OpType() == "reduce_prod" && x_info[0].dtype == P2ODataType::FP64) { - input_name = helper_->AutoCast(x_info[0].name, P2ODataType::FP64, - P2ODataType::FP32); - } - auto reduce_node = helper_->MakeNode(op_map[OpType()], {input_name}); - if (!reduce_all_) { - AddAttribute(reduce_node, "axes", dim_); - } else { - AddAttribute(reduce_node, "axes", Arange(0, x_info[0].Rank())); - } - AddAttribute(reduce_node, "keepdims", static_cast(keep_dim_)); - out = reduce_node->output(0); - if (OpType() == "reduce_prod" && x_info[0].dtype == P2ODataType::FP64) { - out = helper_->AutoCast(reduce_node->output(0), P2ODataType::FP32, - P2ODataType::FP64); - } - } - if (!keep_dim_ && reduce_all_axes) { - out = helper_->Reshape(out, {-1}); - } - helper_->AutoCast(out, out_info[0].name, x_info[0].dtype, out_info[0].dtype); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/reduce.h b/paddle2onnx/mapper/tensor/reduce.h deleted file mode 100755 index 901a3f48c62..00000000000 --- a/paddle2onnx/mapper/tensor/reduce.h +++ /dev/null @@ -1,50 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class ReduceMapper : public Mapper { - public: - ReduceMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - if (OpType() == "logsumexp") { - GetAttr("keepdim", &keep_dim_); - GetAttr("reduce_all", &reduce_all_); - } else { - GetAttr("keep_dim", &keep_dim_); - GetAttr("reduce_all", &reduce_all_); - GetAttr("in_dtype", &in_dtype_); - GetAttr("out_dtype", &out_dtype_); - } - } - void Opset7(); - - int32_t GetMinOpset(bool verbose = false); - - private: - bool keep_dim_; - bool reduce_all_; - int64_t in_dtype_; - int64_t out_dtype_; - std::vector dim_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/reshape2.cc b/paddle2onnx/mapper/tensor/reshape2.cc deleted file mode 100644 index ccdbc5a86e0..00000000000 --- a/paddle2onnx/mapper/tensor/reshape2.cc +++ /dev/null @@ -1,54 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/reshape2.h" - -#include -#include -#include - -namespace paddle2onnx { -REGISTER_MAPPER(reshape2, Reshape2Mapper) - -void Reshape2Mapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - std::string shape_name = "ShapeTensor"; - if (!HasInput(shape_name)) { - shape_name = "Shape"; - } - - std::string new_shape = ""; - if (HasInput(shape_name)) { - auto shape_info = GetInput(shape_name); - if (shape_info.size() > 1) { - new_shape = helper_->ConcatIndices(shape_info); - } else { - new_shape = helper_->AutoCast(shape_info[0].name, shape_info[0].dtype, - P2ODataType::INT64); - } - } else { - std::vector value; - GetAttr("shape", &value); - new_shape = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, value); - } - auto node = helper_->MakeNode("Reshape", {input_info[0].name, new_shape}, - {output_info[0].name}); - if (helper_->GetOpsetVersion()>= 14) { - AddAttribute(node, "allowzero", int64_t(0)); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/reshape2.h b/paddle2onnx/mapper/tensor/reshape2.h deleted file mode 100644 index 65ce4206190..00000000000 --- a/paddle2onnx/mapper/tensor/reshape2.h +++ /dev/null @@ -1,32 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class Reshape2Mapper : public Mapper { - public: - Reshape2Mapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - - void Opset7(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/scale.cc b/paddle2onnx/mapper/tensor/scale.cc deleted file mode 100644 index 810b3472ced..00000000000 --- a/paddle2onnx/mapper/tensor/scale.cc +++ /dev/null @@ -1,76 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. -#include "paddle2onnx/mapper/tensor/scale.h" - -#include - -namespace paddle2onnx { - -REGISTER_MAPPER(scale, ScaleMapper) - -void ScaleMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - bool has_scale_tensor = HasInput("ScaleTensor"); - bool is_scale_1 = ((scale_ - 1.0) < 1e-06 && (scale_ - 1.0) > -1e-06); - bool is_bias_0 = (bias_ < 1e-06 && bias_ > -1e-06); - - if (!has_scale_tensor && is_scale_1 && is_bias_0) { - helper_->MakeNode("Identity", {input_info[0].name}, {output_info[0].name}); - } else { - auto input = helper_->AutoCast(input_info[0].name, input_info[0].dtype, - P2ODataType::FP32); - std::string out = input; - if (bias_after_scale_) { - if (!is_scale_1 || HasInput("ScaleTensor")) { - if (HasInput("ScaleTensor")) { - auto scale_info = GetInput("ScaleTensor"); - auto scale = helper_->AutoCast( - scale_info[0].name, scale_info[0].dtype, P2ODataType::FP32); - out = helper_->MakeNode("Mul", {out, scale})->output(0); - } else { - auto scale = - helper_->Constant({}, ONNX_NAMESPACE::TensorProto::FLOAT, scale_); - out = helper_->MakeNode("Mul", {out, scale})->output(0); - } - } - if (!is_bias_0) { - auto bias = - helper_->Constant({}, ONNX_NAMESPACE::TensorProto::FLOAT, bias_); - out = helper_->MakeNode("Add", {out, bias})->output(0); - } - } else { - if (!is_bias_0) { - auto bias = - helper_->Constant({}, ONNX_NAMESPACE::TensorProto::FLOAT, bias_); - out = helper_->MakeNode("Add", {out, bias})->output(0); - } - if (!is_scale_1 || HasInput("ScaleTensor")) { - if (HasInput("ScaleTensor")) { - auto scale_info = GetInput("ScaleTensor"); - auto scale = helper_->AutoCast( - scale_info[0].name, scale_info[0].dtype, P2ODataType::FP32); - out = helper_->MakeNode("Mul", {out, scale})->output(0); - } else { - auto scale = - helper_->Constant({}, ONNX_NAMESPACE::TensorProto::FLOAT, scale_); - out = helper_->MakeNode("Mul", {out, scale})->output(0); - } - } - } - helper_->AutoCast(out, output_info[0].name, P2ODataType::FP32, - output_info[0].dtype); - } -} -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/scale.h b/paddle2onnx/mapper/tensor/scale.h deleted file mode 100644 index 3970516c3d4..00000000000 --- a/paddle2onnx/mapper/tensor/scale.h +++ /dev/null @@ -1,39 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. -#pragma once -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class ScaleMapper : public Mapper { - public: - ScaleMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("scale", &scale_); - GetAttr("bias", &bias_); - GetAttr("bias_after_scale", &bias_after_scale_); - } - - void Opset7(); - - private: - float scale_ = 1.0; - float bias_ = 0.0; - bool bias_after_scale_ = true; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/scatter.cc b/paddle2onnx/mapper/tensor/scatter.cc deleted file mode 100644 index 74823a2c00b..00000000000 --- a/paddle2onnx/mapper/tensor/scatter.cc +++ /dev/null @@ -1,83 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/scatter.h" - -namespace paddle2onnx { -REGISTER_MAPPER(scatter, ScatterMapper) - -int32_t ScatterMapper::GetMinOpset(bool verbose) { - if (!overwrite_) { - Logger(verbose, 16) << "When overwrite is False, " << RequireOpset(16) - << std::endl; - return 16; - } - Logger(verbose, 11) << RequireOpset(11) << std::endl; - return 11; -} - -void ScatterMapper::Opset11() { - auto input_x_info = GetInput("X"); - auto input_ids_info = GetInput("Ids"); - auto input_updates_info = GetInput("Updates"); - auto output_info = GetOutput("Out"); - - std::string ids_node = helper_->AutoCast( - input_ids_info[0].name, input_ids_info[0].dtype, P2ODataType::INT64); - - std::string shape_node; - if (input_ids_info[0].Rank() == 0) { - std::vector shape = {1}; - shape_node = helper_->Constant(GetOnnxDtype(P2ODataType::INT64), shape); - } else { - std::vector shape = {input_ids_info[0].shape[0], 1}; - shape_node = helper_->Constant(GetOnnxDtype(P2ODataType::INT64), shape); - } - - auto reshape_index_node = - helper_->MakeNode("Reshape", {ids_node, shape_node}); - - if (!overwrite_) { - auto shape_node = helper_->MakeNode("Shape", {input_x_info[0].name}); - std::string zeros_like_node = helper_->ConstOfShape( - shape_node->output(0), GetOnnxDtype(input_x_info[0].dtype), - static_cast(0)); - auto scatter_nd_node = helper_->MakeNode( - "ScatterND", {zeros_like_node, reshape_index_node->output(0), - input_updates_info[0].name}); - AddAttribute(scatter_nd_node, "reduction", "add"); - - std::string zero_node = helper_->Constant( - {}, GetOnnxDtype(input_x_info[0].dtype), static_cast(0)); - - auto equal_node = - helper_->MakeNode("Equal", {scatter_nd_node->output(0), zero_node}); - - std::string condition_node = helper_->AutoCast( - equal_node->output(0), P2ODataType::INT64, P2ODataType::BOOL); - - helper_->MakeNode( - "Where", - {condition_node, input_x_info[0].name, scatter_nd_node->output(0)}, - {output_info[0].name}); - } else { - auto node = - helper_->MakeNode("ScatterND", - {input_x_info[0].name, reshape_index_node->output(0), - input_updates_info[0].name}, - {output_info[0].name}); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/scatter.h b/paddle2onnx/mapper/tensor/scatter.h deleted file mode 100644 index cefa80faeb0..00000000000 --- a/paddle2onnx/mapper/tensor/scatter.h +++ /dev/null @@ -1,37 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class ScatterMapper : public Mapper { - public: - ScatterMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("overwrite", &overwrite_); - } - int32_t GetMinOpset(bool verbose = false); - void Opset11(); - - private: - bool overwrite_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/scatter_nd_add.cc b/paddle2onnx/mapper/tensor/scatter_nd_add.cc deleted file mode 100644 index 7bbe5c92829..00000000000 --- a/paddle2onnx/mapper/tensor/scatter_nd_add.cc +++ /dev/null @@ -1,48 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/scatter_nd_add.h" - -namespace paddle2onnx { -REGISTER_MAPPER(scatter_nd_add, ScatterNdAddMapper) - -int32_t ScatterNdAddMapper::GetMinOpset(bool verbose) { - Logger(verbose, 16) << RequireOpset(16) << std::endl; - return 16; -} - -void ScatterNdAddMapper::Opset16() { - auto input_x_info = GetInput("X"); - auto input_ids_info = GetInput("Index"); - auto input_updates_info = GetInput("Updates"); - auto output_info = GetOutput("Out"); - - auto shape_node = helper_->MakeNode("Shape", {input_x_info[0].name}); - - std::string zeros_like_node = helper_->ConstOfShape( - shape_node->output(0), GetOnnxDtype(input_x_info[0].dtype), - static_cast(0)); - - std::string input_ids_node = helper_->AutoCast( - input_ids_info[0].name, input_ids_info[0].dtype, P2ODataType::INT64); - - auto scatter_nd_node = helper_->MakeNode( - "ScatterND", - {zeros_like_node, input_ids_node, input_updates_info[0].name}); - AddAttribute(scatter_nd_node, "reduction", "add"); - helper_->MakeNode("Add", {input_x_info[0].name, scatter_nd_node->output(0)}, - {output_info[0].name}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/scatter_nd_add.h b/paddle2onnx/mapper/tensor/scatter_nd_add.h deleted file mode 100644 index 053e5076d15..00000000000 --- a/paddle2onnx/mapper/tensor/scatter_nd_add.h +++ /dev/null @@ -1,34 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class ScatterNdAddMapper : public Mapper { - public: - ScatterNdAddMapper(const PaddleParser& p, OnnxHelper* helper, - int64_t block_id, int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - int32_t GetMinOpset(bool verbose = false); - void Opset16(); - - private: -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/set_value.cc b/paddle2onnx/mapper/tensor/set_value.cc deleted file mode 100644 index 7b630b1f958..00000000000 --- a/paddle2onnx/mapper/tensor/set_value.cc +++ /dev/null @@ -1,143 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/set_value.h" - -namespace paddle2onnx { -REGISTER_MAPPER(set_value, SetValueMapper) - -int32_t SetValueMapper::GetMinOpset(bool verbose) { - if (none_axes_.size() > 0) { - Error() << "Attribute none_axes is not supported." << std::endl; - return -1; - } - if (axes_.size() > 1) { - Error() << "Attribute axes is supported while it only contains 1 element." - << std::endl; - return -1; - } - if (steps_.size() > 1) { - Error() << "ttribute steps is supported while it only contains 1 element." - << std::endl; - return -1; - } - if (GetInput("Input")[0].dtype == P2ODataType::BOOL) { - Error() << "Input X with data type of boolean is not supported." - << std::endl; - return -1; - } - Logger(verbose, 12) << RequireOpset(12) << std::endl; - return 12; -} - -void SetValueMapper::Opset12() { - auto input_info = GetInput("Input"); - auto output_info = GetOutput("Out"); - std::string starts = ""; - if (HasInput("StartsTensorList")) { - // if negtive value exists, not supported - starts = helper_->ConcatIndices(GetInput("StartsTensorList")); - } else { - starts = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, starts_); - } - std::string ends = ""; - if (HasInput("EndsTensorList")) { - ends = helper_->ConcatIndices(GetInput("EndsTensorList")); - } else { - // if out of range value in end exists, not supported - ends = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, ends_); - } - - auto input_tensor = input_info[0].name; - std::string axes = helper_->Constant({1}, ONNX_NAMESPACE::TensorProto::INT64, - int64_t(axes_[0])); - // process out of range ends - auto input_shape = helper_->MakeNode("Shape", {input_tensor})->output(0); - auto gather_end_bound = helper_->MakeNode("Gather", {input_shape, axes}); - AddAttribute(gather_end_bound, "axis", int64_t(0)); - ends = - helper_->MakeNode("Min", {gather_end_bound->output(0), ends})->output(0); - - std::string steps = ""; - if (HasInput("StepsTensorList")) { - steps = helper_->ConcatIndices(GetInput("StepsTensorList")); - } else { - steps = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, steps_); - } - std::string value = ""; - int64_t value_rank = input_info[0].Rank(); - if (HasInput("ValueTensor")) { - auto value_info = GetInput("ValueTensor"); - value = value_info[0].name; - value_rank = value_info[0].Rank(); - } else { - value_rank = shape_.size(); - int in_dtype = input_info[0].dtype; - if (in_dtype == P2ODataType::INT32 || in_dtype == P2ODataType::INT64) { - value = helper_->Assign(GetOnnxDtype(output_info[0].dtype), shape_, - int_values_); - } else if (in_dtype == P2ODataType::FP32) { - value = helper_->Assign(GetOnnxDtype(output_info[0].dtype), shape_, - fp32_values_); - } else if (in_dtype == P2ODataType::FP64) { - value = helper_->Assign(GetOnnxDtype(output_info[0].dtype), shape_, - fp64_values_); - } - } - - auto sliced_data = - helper_->MakeNode("Slice", {input_tensor, starts, ends, axes, steps}) - ->output(0); - - auto sliced_shape = helper_->MakeNode("Shape", {sliced_data})->output(0); - if (decrease_axes_.size() > 0 && value_rank != input_info[0].Rank()) { - value = helper_->Unsqueeze(value, decrease_axes_); - } - auto expand_value = - helper_->MakeNode("Expand", {value, sliced_shape})->output(0); - - auto indices = helper_ - ->MakeNode("Range", {helper_->Squeeze(starts, {}), - helper_->Squeeze(ends, {}), - helper_->Squeeze(steps, {})}) - ->output(0); - if (axes_[0] == 0) { - indices = helper_->Unsqueeze(indices, {1}); - helper_->MakeNode("ScatterND", {input_tensor, indices, expand_value}, - {output_info[0].name}); - } else { - std::vector indices_shape(input_info[0].Rank(), 1); - indices_shape[axes_[0]] = -1; - indices = helper_->Reshape(indices, indices_shape); - auto one = - helper_->Constant({1}, ONNX_NAMESPACE::TensorProto::INT64, int64_t(1)); - if (axes_[0] == input_info[0].Rank() - 1) { - auto part_shape = helper_->Slice(sliced_shape, {0}, {0}, {axes_[0]}); - auto tiled_shape = helper_->Concat({part_shape, one}, 0); - indices = helper_->MakeNode("Tile", {indices, tiled_shape})->output(0); - } else { - auto part_0_shape = helper_->Slice(sliced_shape, {0}, {0}, {axes_[0]}); - auto part_1_shape = helper_->Slice(sliced_shape, {0}, {axes_[0] + 1}, - {input_info[0].Rank()}); - auto tiled_shape = helper_->Concat({part_0_shape, one, part_1_shape}, 0); - indices = helper_->MakeNode("Tile", {indices, tiled_shape})->output(0); - } - auto scatter_node = helper_->MakeNode("ScatterElements", - {input_tensor, indices, expand_value}, - {output_info[0].name}); - AddAttribute(scatter_node, "axis", axes_[0]); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/set_value.h b/paddle2onnx/mapper/tensor/set_value.h deleted file mode 100644 index c5ddb9a97a1..00000000000 --- a/paddle2onnx/mapper/tensor/set_value.h +++ /dev/null @@ -1,65 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class SetValueMapper : public Mapper { - public: - SetValueMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - MarkAsExperimentalOp(); - GetAttr("axes", &axes_); - GetAttr("starts", &starts_); - GetAttr("ends", &ends_); - GetAttr("steps", &steps_); - GetAttr("shape", &shape_); - GetAttr("decrease_axes", &decrease_axes_); - GetAttr("none_axes", &none_axes_); - if (!HasInput("ValueTensor")) { - auto dtype = GetInput("Input")[0].dtype; - if (dtype == P2ODataType::INT32) { - GetAttr("int32_values", &int_values_); - } else if (dtype == P2ODataType::INT64) { - GetAttr("int64_values", &int_values_); - } else if (dtype == P2ODataType::FP32) { - GetAttr("fp32_values", &fp32_values_); - } else if (dtype == P2ODataType::FP64) { - GetAttr("fp64_values", &fp64_values_); - } - } - } - int32_t GetMinOpset(bool verbose = false); - void Opset12(); - - private: - std::vector axes_; - std::vector starts_; - std::vector ends_; - std::vector steps_; - std::vector shape_; - std::vector decrease_axes_; - std::vector none_axes_; - std::vector int_values_; - std::vector fp32_values_; - std::vector fp64_values_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/slice.cc b/paddle2onnx/mapper/tensor/slice.cc deleted file mode 100644 index 317f3b9d5a7..00000000000 --- a/paddle2onnx/mapper/tensor/slice.cc +++ /dev/null @@ -1,175 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/slice.h" - -#include -#include -#include - -namespace paddle2onnx { -REGISTER_MAPPER(slice, SliceMapper) -REGISTER_MAPPER(strided_slice, SliceMapper) - -int32_t SliceMapper::GetMinOpset(bool verbose) { - if (HasInput("StartsTensorList") || HasInput("EndsTensorList") || - HasInput("StridesTensorList")) { - Logger(verbose, 10) - << "While has input StartsTensorList/EndsTensorListStridesTensorList, " - << RequireOpset(10) << std::endl; - return 10; - } - if (HasInput("StartsTensor")) { - auto info = GetInput("StartsTensor"); - if (!IsConstantInput("StartsTensor")) { - Logger(verbose, 10) - << "While has input StartsTensor, and it's not a constant tensor, " - << RequireOpset(10) << std::endl; - return 10; - } - } - if (HasInput("EndsTensor")) { - auto info = GetInput("EndsTensor"); - if (!IsConstantInput("EndsTensor")) { - Logger(verbose, 10) - << "While has input EndsTensor, and it's not a constant tensor, " - << RequireOpset(10) << std::endl; - return 10; - } - } - if (HasInput("StridesTensor") || strides_.size() > 0) { - Logger(verbose, 10) << "While has strides, " << RequireOpset(10) - << std::endl; - return 10; - } - return 7; -} - -std::vector SliceMapper::DecreaseAxis() { - std::vector decrease_axis; - bool has_attr = HasAttr("decrease_axis"); - if (has_attr) { - GetAttr("decrease_axis", &decrease_axis); - auto input_info = GetInput("Input"); - auto output_info = GetOutput("Out"); - if (output_info[0].shape.size() == 1 && output_info[0].shape[0] == 0) { - return decrease_axis; - } - if (input_info[0].shape.size() > output_info[0].shape.size()) { - return decrease_axis; - } - return {}; - } - return decrease_axis; -} - -void SliceMapper::Opset7() { - auto input_info = GetInput("Input"); - auto output_info = GetOutput("Out"); - - Assert(!HasInput("StartsTensorList"), - "While slice/strided_slice has input StartsTensorList, requires " - "opset_version >= 10"); - - std::vector starts; - if (HasInput("StartsTensor")) { - Assert(TryGetInputValue("StartsTensor", &starts), - "While slice/strided_slice has input StartsTensor, and it's not a " - "constant tensor, then requires opset_version >= 10"); - } else { - starts = starts_; - } - - Assert(!HasInput("EndsTensorList"), - "While slice/strided_slice has input EndsTensorList, requires " - "opset_version >= 10"); - std::vector ends; - if (HasInput("EndsTensor")) { - auto info = GetInput("EndsTensor"); - Assert(TryGetInputValue("EndsTensor", &ends), - "While slice/strided_slice has input EndsTensor, and it's not a " - "constant tensor, then requires opset_version >= 10"); - } else { - ends = ends_; - } - - std::vector decrease_axis = DecreaseAxis(); - if (decrease_axis.empty()) { - helper_->Slice(input_info[0].name, output_info[0].name, axes_, starts, - ends); - } else { - std::string node = helper_->Slice(input_info[0].name, axes_, starts, ends); - helper_->Squeeze(node, output_info[0].name, decrease_axis); - } -} - -void SliceMapper::Opset10() { - auto input_info = GetInput("Input"); - auto output_info = GetOutput("Out"); - - std::string starts = ""; - if (HasInput("StartsTensorList")) { - auto info = GetInput("StartsTensorList"); - starts = helper_->ConcatIndices(info); - } else if (HasInput("StartsTensor")) { - auto info = GetInput("StartsTensor"); - starts = helper_->AutoCast(info[0].name, info[0].dtype, P2ODataType::INT64); - } else { - starts = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, starts_); - } - - std::string ends = ""; - if (HasInput("EndsTensorList")) { - auto info = GetInput("EndsTensorList"); - ends = helper_->ConcatIndices(info); - } else if (HasInput("EndsTensor")) { - auto info = GetInput("EndsTensor"); - ends = helper_->AutoCast(info[0].name, info[0].dtype, P2ODataType::INT64); - } else { - ends = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, ends_); - } - - std::string strides = ""; - if (HasInput("StridesTensorList")) { - auto info = GetInput("StridesTensorList"); - strides = helper_->ConcatIndices(info); - } else if (HasInput("StridesTensor")) { - auto info = GetInput("StridesTensor"); - strides = - helper_->AutoCast(info[0].name, info[0].dtype, P2ODataType::INT64); - } else { - if (strides_.size() == 0) { - strides = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, - std::vector(axes_.size(), 1)); - } else { - strides = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, strides_); - } - } - - auto axes = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, axes_); - std::vector decrease_axis = DecreaseAxis(); - if (decrease_axis.empty()) { - helper_->MakeNode("Slice", - {input_info[0].name, starts, ends, axes, strides}, - {output_info[0].name}); - } else { - auto out = helper_ - ->MakeNode("Slice", - {input_info[0].name, starts, ends, axes, strides}) - ->output(0); - helper_->Squeeze(out, output_info[0].name, decrease_axis); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/slice.h b/paddle2onnx/mapper/tensor/slice.h deleted file mode 100644 index 38e5dc2f73c..00000000000 --- a/paddle2onnx/mapper/tensor/slice.h +++ /dev/null @@ -1,52 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class SliceMapper : public Mapper { - public: - SliceMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("axes", &axes_); - GetAttr("starts", &starts_); - GetAttr("ends", &ends_); - if (HasAttr("strides")) { - GetAttr("strides", &strides_); - } - if (HasAttr("decrease_axis_")) { - GetAttr("decrease_axis", &decrease_axis_); - } - } - - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - void Opset10(); - - private: - std::vector DecreaseAxis(); - std::vector axes_; - std::vector starts_; - std::vector ends_; - std::vector strides_; - std::vector decrease_axis_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/split.cc b/paddle2onnx/mapper/tensor/split.cc deleted file mode 100644 index 2642c85b668..00000000000 --- a/paddle2onnx/mapper/tensor/split.cc +++ /dev/null @@ -1,145 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/split.h" - -namespace paddle2onnx { -REGISTER_MAPPER(split, SplitMapper) - -int32_t SplitMapper::GetMinOpset(bool verbose) { - int64_t axis = axis_; - if (HasInput("AxisTensor")) { - std::vector value; - if (!TryGetInputValue("AxisTensor", &value)) { - Error() << "While AxisTensor as the input and it's not a constant " - "tensor, the conversion is not supported yet." - << std::endl; - return -1; - } - axis = value[0]; - } - - if (HasInput("SectionsTensorList")) { - Logger(verbose, 13) << "While has input SectionsTensorList, " - << RequireOpset(13) << std::endl; - return 13; - } - - for (size_t i = 0; i < sections_.size(); ++i) { - if (sections_[i] < 0) { - auto info = GetInput("X"); - if (info[0].shape[axis] < 0) { - Error() << "Cannot convert split op, while there's -1 in sections and " - "cannot be infered by input shape." - << std::endl; - return -1; - } - } - } - return 7; -} - -void SplitMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - int64_t axis = axis_; - if (HasInput("AxisTensor")) { - std::vector value; - Assert(TryGetInputValue("AxisTensor", &value), - "[Paddle2ONNX](split) Cannot get constant value from AxisTensor."); - axis = value[0]; - } - if (axis < 0) { - axis += input_info[0].Rank(); - } - Assert(!HasInput("SectionsTensorList"), - "[Paddle2ONNX](split) While SectionTensorList as input, requires " - "opset_version >= 13."); - - int sum_of_kown_dim = 0; - for (size_t i = 0; i < sections_.size(); ++i) { - if (sections_[i] > 0) { - sum_of_kown_dim += sections_[i]; - } - } - for (size_t i = 0; i < sections_.size(); ++i) { - if (sections_[i] < 0) { - Assert(input_info[0].shape[axis] > 0, - "Cannot convert split op, while there's -1 in sections and cannot " - "be infered by input shape."); - sections_[i] = input_info[0].shape[axis] - sum_of_kown_dim; - } - } - - std::vector output_names(output_info.size()); - for (size_t i = 0; i < output_info.size(); ++i) { - output_names[i] = output_info[i].name; - } - - helper_->Split(input_info[0].name, output_names, sections_, axis); -} - -void SplitMapper::Opset13() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - int64_t axis = axis_; - if (HasInput("AxisTensor")) { - std::vector value; - Assert(TryGetInputValue("AxisTensor", &value), - "[Paddle2ONNX](split) Cannot get constant value from AxisTensor."); - axis = value[0]; - } - if (axis < 0) { - axis += input_info[0].Rank(); - } - - std::string splits = ""; - if (HasInput("SectionsTensorList")) { - auto info = GetInput("SectionsTensorList"); - splits = helper_->ConcatIndices(info); - } else if (sections_.size() > 0) { - int sum_of_kown_dim = 0; - for (size_t i = 0; i < sections_.size(); ++i) { - if (sections_[i] > 0) { - sum_of_kown_dim += sections_[i]; - } - } - for (size_t i = 0; i < sections_.size(); ++i) { - if (sections_[i] < 0) { - Assert(input_info[0].shape[axis] > 0, - "Cannot convert split op, while there's -1 in sections and " - "cannot be infered by input shape."); - sections_[i] = input_info[0].shape[axis] - sum_of_kown_dim; - } - } - splits = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, sections_); - } - - std::vector output_names(output_info.size()); - for (size_t i = 0; i < output_info.size(); ++i) { - output_names[i] = output_info[i].name; - } - if (splits != "") { - auto node = - helper_->MakeNode("Split", {input_info[0].name, splits}, output_names); - AddAttribute(node, "axis", axis); - } else { - auto node = helper_->MakeNode("Split", {input_info[0].name}, output_names); - AddAttribute(node, "axis", axis); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/split.h b/paddle2onnx/mapper/tensor/split.h deleted file mode 100644 index 2665cfcdb42..00000000000 --- a/paddle2onnx/mapper/tensor/split.h +++ /dev/null @@ -1,41 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class SplitMapper : public Mapper { - public: - SplitMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("axis", &axis_); - GetAttr("sections", §ions_); - } - - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - void Opset13(); - - private: - int64_t axis_; - std::vector sections_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/squeeze2.cc b/paddle2onnx/mapper/tensor/squeeze2.cc deleted file mode 100644 index 953995875a9..00000000000 --- a/paddle2onnx/mapper/tensor/squeeze2.cc +++ /dev/null @@ -1,84 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/squeeze2.h" - -namespace paddle2onnx { -REGISTER_MAPPER(squeeze2, Squeeze2Mapper) - -int32_t Squeeze2Mapper::GetMinOpset(bool verbose) { - if (IsAttrVar("axes")) { - auto infos = GetAttrVar("axes"); - for (auto &info : infos) { - if (!IsConstant(info)) { - return 13; - } - } - } - return 7; -} - -void Squeeze2Mapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - std::vector ret; - ret.reserve(input_info[0].shape.size()); - for (auto i : input_info[0].shape) { - if (i > 1) ret.push_back(i); - } - if (ret.size() == input_info[0].Rank()) { - helper_->MakeNode("Identity", {input_info[0].name}, {output_info[0].name}); - } else { - if (helper_->GetOpsetVersion() >= 13 && IsAttrVar("axes")) { - auto axes_info = GetAttrVar("axes"); - std::string axes_name; - if (axes_info.size() == 1U) { - axes_name = helper_->AutoCast(axes_info[0].name, axes_info[0].dtype, - P2ODataType::INT64); - } else { - axes_name = helper_->ConcatIndices(axes_info); - } - helper_->MakeNode("Squeeze", {input_info[0].name, axes_name}, - {output_info[0].name}); - } else { - if (IsAttrVar("axes")) { - auto axes_info = GetAttrVar("axes"); - for (int64_t index = 0; index < axes_info.size(); index++) { - std::vector temp; - TryGetValue(axes_info[index], &temp); - for (auto &data : temp) { - axes_.push_back(data); - } - } - } else { - GetAttr("axes", &axes_); - } - std::vector axes(axes_.begin(), axes_.end()); - for (size_t i = 0; i < axes.size(); ++i) { - if (axes[i] < 0) { - axes[i] += input_info[0].Rank(); - } - } - if (axes.size() > 0) { - std::sort(axes.begin(), axes.end()); - helper_->Squeeze(input_info[0].name, output_info[0].name, axes); - } else { - helper_->Squeeze(input_info[0].name, output_info[0].name, {}); - } - } - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/squeeze2.h b/paddle2onnx/mapper/tensor/squeeze2.h deleted file mode 100644 index ffa31c88546..00000000000 --- a/paddle2onnx/mapper/tensor/squeeze2.h +++ /dev/null @@ -1,35 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class Squeeze2Mapper : public Mapper { - public: - Squeeze2Mapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - std::vector axes_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/stack.cc b/paddle2onnx/mapper/tensor/stack.cc deleted file mode 100644 index 4c6f249db1a..00000000000 --- a/paddle2onnx/mapper/tensor/stack.cc +++ /dev/null @@ -1,39 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/stack.h" - -namespace paddle2onnx { -REGISTER_MAPPER(stack, StackMapper) - -void StackMapper::Opset7() { - auto x_info = GetInput("X"); - auto y_info = GetOutput("Y"); - - int32_t out_dtype = 0; - std::vector aligned_inputs = - helper_->DtypeAlignment(x_info, &out_dtype); - auto axis = axis_; - if (axis < 0) { - axis = axis + x_info[0].Rank() + 1; - } - for (size_t i = 0; i < aligned_inputs.size(); ++i) { - aligned_inputs[i] = - helper_->Unsqueeze(aligned_inputs[i], std::vector(1, axis)); - } - auto out = helper_->Concat(aligned_inputs, axis_); - helper_->AutoCast(out, y_info[0].name, out_dtype, y_info[0].dtype); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/stack.h b/paddle2onnx/mapper/tensor/stack.h deleted file mode 100644 index 36d01d54a55..00000000000 --- a/paddle2onnx/mapper/tensor/stack.h +++ /dev/null @@ -1,33 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class StackMapper : public Mapper { - public: - StackMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("axis", &axis_); - } - void Opset7(); - - private: - int64_t axis_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/take_along_axis.cc b/paddle2onnx/mapper/tensor/take_along_axis.cc deleted file mode 100755 index 7262ffb3f72..00000000000 --- a/paddle2onnx/mapper/tensor/take_along_axis.cc +++ /dev/null @@ -1,38 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/take_along_axis.h" - -namespace paddle2onnx { -REGISTER_MAPPER(take_along_axis, TakeAlongAxisMapper) - -int32_t TakeAlongAxisMapper::GetMinOpset(bool verbose) { - Logger(verbose, 11) << RequireOpset(11) << std::endl; - return 11; -} - -void TakeAlongAxisMapper::Opset11() { - auto x_info = GetInput("Input"); - auto index_info = GetInput("Index"); - auto out_info = GetOutput("Result"); - - auto axis = axis_; - - auto node = - helper_->MakeNode("GatherElements", {x_info[0].name, index_info[0].name}, - {out_info[0].name}); - AddAttribute(node, "axis", axis); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/take_along_axis.h b/paddle2onnx/mapper/tensor/take_along_axis.h deleted file mode 100755 index 8299d3de926..00000000000 --- a/paddle2onnx/mapper/tensor/take_along_axis.h +++ /dev/null @@ -1,38 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class TakeAlongAxisMapper : public Mapper { - public: - TakeAlongAxisMapper(const PaddleParser& p, OnnxHelper* helper, - int64_t block_id, int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("Axis", &axis_); - } - - int32_t GetMinOpset(bool verbose = false); - void Opset11(); - - private: - int64_t axis_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/temporal_shift.cc b/paddle2onnx/mapper/tensor/temporal_shift.cc deleted file mode 100644 index 2a2aac51066..00000000000 --- a/paddle2onnx/mapper/tensor/temporal_shift.cc +++ /dev/null @@ -1,79 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/temporal_shift.h" - -namespace paddle2onnx { -REGISTER_MAPPER(temporal_shift, TemporalShiftMapper) - -int32_t TemporalShiftMapper::GetMinOpset(bool verbose) { - if (data_format_ == "NHWC") { - Error() << "Only support data_format of NCHW, but now the data format is " - << data_format_ << "." << std::endl; - return -1; - } - auto input_info = GetOutput("Out"); - if (input_info[0].Rank() != 4) { - Error() << "The input dims must be 4, but now the input dims is " - << std::to_string(input_info[0].Rank()) << "." << std::endl; - return -1; - } - return 7; -} - -void TemporalShiftMapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - int64_t C = input_info[0].shape[1]; - int64_t H = input_info[0].shape[2]; - int64_t W = input_info[0].shape[3]; - std::vector reshape_shape = {-1, seg_num_, C, H, W}; - - std::string reshape_input = - helper_->Reshape(input_info[0].name, reshape_shape); - - std::vector paddings(10, 0); - paddings[1] = 1; - paddings[6] = 1; - - std::string padding_constant_node = - helper_->Constant(GetOnnxDtype(P2ODataType::INT64), paddings); - - std::string pad_node = ""; - if (helper_->GetOpsetVersion() < 11) { - auto node = helper_->MakeNode("Pad", {reshape_input}); - AddAttribute(node, "pads", paddings); - float val = 0.0; - AddAttribute(node, "value", val); - pad_node = node->output(0); - } else { - auto node = - helper_->MakeNode("Pad", {reshape_input, padding_constant_node}); - pad_node = node->output(0); - } - - int64_t C1 = C * shift_ratio_; - int64_t C2 = 2 * C * shift_ratio_; - std::string slice_1 = - helper_->Slice(pad_node, {1, 2}, {0, 0}, {seg_num_, C1}); - std::string slice_2 = - helper_->Slice(pad_node, {1, 2}, {2, C1}, {2 + seg_num_, C2}); - std::string slice_3 = - helper_->Slice(pad_node, {1, 2}, {1, C2}, {1 + seg_num_, C}); - std::string concat_out = helper_->Concat({slice_1, slice_2, slice_3}, 2); - helper_->Reshape(concat_out, output_info[0].name, {-1, C, H, W}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/temporal_shift.h b/paddle2onnx/mapper/tensor/temporal_shift.h deleted file mode 100644 index d01ffb2718a..00000000000 --- a/paddle2onnx/mapper/tensor/temporal_shift.h +++ /dev/null @@ -1,41 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class TemporalShiftMapper : public Mapper { - public: - TemporalShiftMapper(const PaddleParser& p, OnnxHelper* helper, - int64_t block_id, int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("data_format", &data_format_); - GetAttr("shift_ratio", &shift_ratio_); - GetAttr("seg_num", &seg_num_); - } - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - - private: - int64_t seg_num_; - float shift_ratio_; - std::string data_format_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/tile.cc b/paddle2onnx/mapper/tensor/tile.cc deleted file mode 100644 index 990775da960..00000000000 --- a/paddle2onnx/mapper/tensor/tile.cc +++ /dev/null @@ -1,60 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/tile.h" - -namespace paddle2onnx { -REGISTER_MAPPER(tile, TileMapper) - -void TileMapper::Opset7() { - auto x_info = GetInput("X"); - auto out_info = GetOutput("Out"); - - bool has_repeats_tensor = HasInput("RepeatTimes"); - bool has_repeats_tensor_list = HasInput("repeat_times_tensor"); - std::string repeats = ""; - // NOTE(Aurelius84): we need to deprecate this branch in the future. - if (has_repeats_tensor) { - auto repeats_info = GetInput("RepeatTimes"); - repeats = helper_->AutoCast(repeats_info[0].name, repeats_info[0].dtype, - P2ODataType::INT64); - } else if (has_repeats_tensor_list) { - auto repeats_info = GetInput("repeat_times_tensor"); - repeats = helper_->ConcatIndices(repeats_info); - } else if (IsAttrVar("repeat_times")) { - auto repeats_info = GetAttrVar("repeat_times"); - if (repeats_info.size() == 1U) { // repeat_times is a whole Tensor - repeats = helper_->AutoCast(repeats_info[0].name, repeats_info[0].dtype, - P2ODataType::INT64); - } else { // repeat_times is a tensor list - repeats = helper_->ConcatIndices(repeats_info); - } - } else { - std::vector values; - GetAttr("repeat_times", &values); - int64_t nums = values.size(); - for (int64_t i = 0; i < x_info[0].Rank() - nums; i++) { - values.insert(values.begin(), 1); - } - repeats = helper_->Constant(ONNX_NAMESPACE::TensorProto::INT64, values); - } - if (x_info[0].Rank() == 0) { - auto unsqueeze = helper_->Unsqueeze(x_info[0].name, {0}); - helper_->MakeNode("Tile", {unsqueeze, repeats}, {out_info[0].name}); - } else { - helper_->MakeNode("Tile", {x_info[0].name, repeats}, {out_info[0].name}); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/tile.h b/paddle2onnx/mapper/tensor/tile.h deleted file mode 100644 index f8a55494dd9..00000000000 --- a/paddle2onnx/mapper/tensor/tile.h +++ /dev/null @@ -1,28 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class TileMapper : public Mapper { - public: - TileMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/top_k.cc b/paddle2onnx/mapper/tensor/top_k.cc deleted file mode 100644 index 0075c9bab57..00000000000 --- a/paddle2onnx/mapper/tensor/top_k.cc +++ /dev/null @@ -1,49 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/top_k.h" - -namespace paddle2onnx { -REGISTER_MAPPER(top_k, TopKMapper) - -void TopKMapper::Opset11() { - auto x_info = GetInput("X"); - auto output_info = GetOutput("Out"); - auto indices_info = GetOutput("Indices"); - if (x_info[0].Rank() == 0) { - helper_->MakeNode("Identity", {x_info[0].name}, {output_info[0].name}); - helper_->Constant(indices_info[0].name, {}, - ONNX_NAMESPACE::TensorProto::INT64, 0); - return; - } - std::string k = ""; - if (HasInput("K")) { - auto k_info = GetInput("K"); - k = helper_->AutoCast(k_info[0].name, k_info[0].dtype, P2ODataType::INT64); - if (k_info[0].Rank() == 0) { - k = helper_->Reshape(k, std::vector(1, -1)); - } - } else { - int64_t k_value = 0; - GetAttr("k", &k_value); - k = helper_->Constant({1}, ONNX_NAMESPACE::TensorProto::INT64, k_value); - } - auto out_node = helper_->MakeNode("TopK", {x_info[0].name, k}, 2); - helper_->AutoCast(out_node->output(0), output_info[0].name, x_info[0].dtype, - output_info[0].dtype); - helper_->AutoCast(out_node->output(1), indices_info[0].name, - P2ODataType::INT64, indices_info[0].dtype); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/top_k.h b/paddle2onnx/mapper/tensor/top_k.h deleted file mode 100644 index 868c6b35677..00000000000 --- a/paddle2onnx/mapper/tensor/top_k.h +++ /dev/null @@ -1,32 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class TopKMapper : public Mapper { - public: - TopKMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - int32_t GetMinOpset(bool verbose) { - Logger(verbose, 11) << RequireOpset(11) << std::endl; - return 11; - } - void Opset11(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/top_k_v2.cc b/paddle2onnx/mapper/tensor/top_k_v2.cc deleted file mode 100644 index 9ab87a8eddc..00000000000 --- a/paddle2onnx/mapper/tensor/top_k_v2.cc +++ /dev/null @@ -1,52 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/top_k_v2.h" - -namespace paddle2onnx { -REGISTER_MAPPER(top_k_v2, TopKV2Mapper) - -void TopKV2Mapper::Opset11() { - auto x_info = GetInput("X"); - auto output_info = GetOutput("Out"); - auto indices_info = GetOutput("Indices"); - if (x_info[0].Rank() == 0) { - helper_->MakeNode("Identity", {x_info[0].name}, {output_info[0].name}); - helper_->Constant(indices_info[0].name, {}, - ONNX_NAMESPACE::TensorProto::INT64, 0); - return; - } - std::string k = ""; - if (HasInput("K")) { - auto k_info = GetInput("K"); - k = helper_->AutoCast(k_info[0].name, k_info[0].dtype, P2ODataType::INT64); - if (k_info[0].Rank() == 0) { - k = helper_->Reshape(k, std::vector(1, -1)); - } - } else { - int64_t k_value = 0; - GetAttr("k", &k_value); - k = helper_->Constant({1}, ONNX_NAMESPACE::TensorProto::INT64, k_value); - } - auto out_node = helper_->MakeNode("TopK", {x_info[0].name, k}, 2); - AddAttribute(out_node, "largest", static_cast(largest_)); - AddAttribute(out_node, "sorted", static_cast(sorted_)); - AddAttribute(out_node, "axis", axis_); - helper_->AutoCast(out_node->output(0), output_info[0].name, x_info[0].dtype, - output_info[0].dtype); - helper_->AutoCast(out_node->output(1), indices_info[0].name, - P2ODataType::INT64, indices_info[0].dtype); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/top_k_v2.h b/paddle2onnx/mapper/tensor/top_k_v2.h deleted file mode 100644 index 1610e28e079..00000000000 --- a/paddle2onnx/mapper/tensor/top_k_v2.h +++ /dev/null @@ -1,41 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class TopKV2Mapper : public Mapper { - public: - TopKV2Mapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("largest", &largest_); - GetAttr("sorted", &sorted_); - GetAttr("axis", &axis_); - } - int32_t GetMinOpset(bool verbose) { - Logger(verbose, 11) << RequireOpset(11) << std::endl; - return 11; - } - void Opset11(); - - private: - bool largest_; - bool sorted_; - int64_t axis_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/transpose2.cc b/paddle2onnx/mapper/tensor/transpose2.cc deleted file mode 100644 index 8dc3ea25762..00000000000 --- a/paddle2onnx/mapper/tensor/transpose2.cc +++ /dev/null @@ -1,36 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/transpose2.h" - -#include -#include - -namespace paddle2onnx { -REGISTER_MAPPER(transpose2, Transpose2Mapper) - -void Transpose2Mapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - if (input_info[0].Rank() == 0) { - helper_->MakeNode("Identity", {input_info[0].name}, {output_info[0].name}); - return; - } - GetAttr("axis", &axis_); - auto node = helper_->MakeNode("Transpose", {input_info[0].name}, - {output_info[0].name}); - AddAttribute(node, "perm", axis_); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/transpose2.h b/paddle2onnx/mapper/tensor/transpose2.h deleted file mode 100644 index 44110e04130..00000000000 --- a/paddle2onnx/mapper/tensor/transpose2.h +++ /dev/null @@ -1,34 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class Transpose2Mapper : public Mapper { - public: - Transpose2Mapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - void Opset7(); - - private: - std::vector axis_ = {}; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/unique.cc b/paddle2onnx/mapper/tensor/unique.cc deleted file mode 100644 index 38e5d1e6baf..00000000000 --- a/paddle2onnx/mapper/tensor/unique.cc +++ /dev/null @@ -1,49 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/unique.h" - -namespace paddle2onnx { -REGISTER_MAPPER(unique, UniqueMapper) - -int32_t UniqueMapper::GetMinOpset(bool verbose) { - Logger(verbose, 11) << RequireOpset(11) << std::endl; - return 11; -} - -void UniqueMapper::Opset11() { - auto intput_info = GetInput("X"); - auto output_out_info = GetOutput("Out"); - auto output_indices_info = GetOutput("Indices"); - auto output_index_info = GetOutput("Index"); - auto output_counts_info = GetOutput("Counts"); - - std::string out_index_name = "out_index"; - std::string out_indices_name = "out_indices"; - std::string out_counts_name = "out_counts"; - auto node = helper_->MakeNode("Unique", {intput_info[0].name}, - {output_out_info[0].name, out_indices_name, - out_index_name, out_counts_name}); - if (axis_.size()) { - AddAttribute(node, "axis", axis_[0]); - } - helper_->AutoCast(out_index_name, output_index_info[0].name, - P2ODataType::INT64, output_index_info[0].dtype); - helper_->AutoCast(out_indices_name, output_indices_info[0].name, - P2ODataType::INT64, output_indices_info[0].dtype); - helper_->AutoCast(out_counts_name, output_counts_info[0].name, - P2ODataType::INT64, output_counts_info[0].dtype); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/unique.h b/paddle2onnx/mapper/tensor/unique.h deleted file mode 100644 index 1049c5c97fa..00000000000 --- a/paddle2onnx/mapper/tensor/unique.h +++ /dev/null @@ -1,39 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class UniqueMapper : public Mapper { - public: - UniqueMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("axis", &axis_); - GetAttr("dtype", &dtype_); - } - int32_t GetMinOpset(bool verbose = false); - void Opset11(); - - private: - std::vector axis_; - int64_t dtype_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/unsqueeze2.cc b/paddle2onnx/mapper/tensor/unsqueeze2.cc deleted file mode 100644 index ddeba0a1df9..00000000000 --- a/paddle2onnx/mapper/tensor/unsqueeze2.cc +++ /dev/null @@ -1,94 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/unsqueeze2.h" - -namespace paddle2onnx { -REGISTER_MAPPER(unsqueeze2, Unsqueeze2Mapper) - -int32_t Unsqueeze2Mapper::GetMinOpset(bool verbose) { - if (axes_.size() == 0) { - if (HasInput("AxesTensorList")) { - Logger(verbose, 13) << "While AxisTensorList as input, " - << RequireOpset(13) << std::endl; - return 13; - } else if (HasInput("AxesTensor")) { - auto info = GetInput("AxesTensor"); - if (!IsConstantInput("AxesTensor")) { - Logger(verbose, 13) - << "While AxesTensor as input, and it's not a constant tensor, " - << RequireOpset(13) << std::endl; - return 13; - } else { - return 7; - } - } - } - return 7; -} - -void Unsqueeze2Mapper::Opset7() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - std::vector axes; - if (axes_.empty()) { - Assert(TryGetInputValue("AxesTensor", &axes), - "While unsqueeze2 has input AxesTensor, it cannot be exported by " - "Paddle2ONNX"); - } else { - axes.assign(axes_.begin(), axes_.end()); - } - for (size_t i = 0; i < axes.size(); ++i) { - if (axes[i] < 0) { - axes[i] = axes[i] + input_info[0].Rank() + i + 1; - } - } - helper_->Unsqueeze(input_info[0].name, output_info[0].name, axes); -} - -void Unsqueeze2Mapper::Opset13() { - auto input_info = GetInput("X"); - auto output_info = GetOutput("Out"); - - std::vector axes; - if (axes_.empty()) { - TryGetInputValue("AxesTensor", &axes); - } else { - axes.assign(axes_.begin(), axes_.end()); - } - for (size_t i = 0; i < axes.size(); ++i) { - if (axes[i] < 0) { - axes[i] = axes[i] + input_info[0].Rank() + i + 1; - } - } - - if (axes.size() > 0) { - helper_->Unsqueeze(input_info[0].name, output_info[0].name, axes); - } else { - std::string axes_node = ""; - if (HasInput("AxesTensorList")) { - auto info = GetInput("AxesTensorList"); - axes_node = helper_->ConcatIndices(info); - } else { - auto info = GetInput("AxesTensor"); - axes_node = - helper_->AutoCast(info[0].name, info[0].dtype, P2ODataType::INT64); - } - helper_->MakeNode("Unsqueeze", {input_info[0].name, axes_node}, - {output_info[0].name}); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/unsqueeze2.h b/paddle2onnx/mapper/tensor/unsqueeze2.h deleted file mode 100644 index 1d1fe2771d7..00000000000 --- a/paddle2onnx/mapper/tensor/unsqueeze2.h +++ /dev/null @@ -1,38 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class Unsqueeze2Mapper : public Mapper { - public: - Unsqueeze2Mapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("axes", &axes_); - } - int32_t GetMinOpset(bool verbose = false); - void Opset7(); - void Opset13(); - - private: - std::vector axes_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/unstack.cc b/paddle2onnx/mapper/tensor/unstack.cc deleted file mode 100644 index 832dc94a3a1..00000000000 --- a/paddle2onnx/mapper/tensor/unstack.cc +++ /dev/null @@ -1,36 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/unstack.h" - -namespace paddle2onnx { -REGISTER_MAPPER(unstack, UnstackMapper) - -void UnstackMapper::Opset7() { - auto x_info = GetInput("X"); - auto y_info = GetOutput("Y"); - - if (axis_ < 0) { - axis_ = axis_ + x_info[0].Rank(); - } - - auto split_nodes = helper_->Split( - x_info[0].name, std::vector(y_info.size(), 1), axis_); - - for (size_t i = 0; i < split_nodes.size(); ++i) { - helper_->Squeeze(split_nodes[i], y_info[i].name, {axis_}); - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/unstack.h b/paddle2onnx/mapper/tensor/unstack.h deleted file mode 100644 index 1ad3d6f0c5e..00000000000 --- a/paddle2onnx/mapper/tensor/unstack.h +++ /dev/null @@ -1,33 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class UnstackMapper : public Mapper { - public: - UnstackMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) { - GetAttr("axis", &axis_); - } - void Opset7(); - - private: - int64_t axis_; -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/where.cc b/paddle2onnx/mapper/tensor/where.cc deleted file mode 100644 index f8adf161493..00000000000 --- a/paddle2onnx/mapper/tensor/where.cc +++ /dev/null @@ -1,31 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/mapper/tensor/where.h" - -namespace paddle2onnx { -REGISTER_MAPPER(where, WhereMapper) - -void WhereMapper::Opset9() { - auto x_info = GetInput("X"); - auto y_info = GetInput("Y"); - auto cond_info = GetInput("Condition"); - auto out_info = GetOutput("Out"); - - helper_->MakeNode("Where", - {cond_info[0].name, x_info[0].name, y_info[0].name}, - {out_info[0].name}); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/mapper/tensor/where.h b/paddle2onnx/mapper/tensor/where.h deleted file mode 100644 index 5f64303f7a0..00000000000 --- a/paddle2onnx/mapper/tensor/where.h +++ /dev/null @@ -1,36 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include "paddle2onnx/mapper/mapper.h" - -namespace paddle2onnx { - -class WhereMapper : public Mapper { - public: - WhereMapper(const PaddleParser& p, OnnxHelper* helper, int64_t block_id, - int64_t op_id) - : Mapper(p, helper, block_id, op_id) {} - - int32_t GetMinOpset(bool verbose = false) { - Logger(verbose, 9) << RequireOpset(9) << std::endl; - return 9; - } - void Opset9(); -}; - -} // namespace paddle2onnx diff --git a/paddle2onnx/mappers_registry.h.in b/paddle2onnx/mappers_registry.h.in deleted file mode 100755 index 5387f8ab1af..00000000000 --- a/paddle2onnx/mappers_registry.h.in +++ /dev/null @@ -1,214 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. -#pragma once - -// If you want to use the paddle2onnx static library, add the -// registration logic here synchronously each time you add a new op. -#ifndef WITH_PADDLE2ONNX_STATIC_INTERNAL -#cmakedefine WITH_PADDLE2ONNX_STATIC_INTERNAL -#endif - -#if defined(WITH_PADDLE2ONNX_STATIC_INTERNAL) -#if (!defined(WITH_PADDLE2ONNX_STATIC_INTERNAL_AT_COMPILING)) - -#if defined(_WIN32) -#define UNUSED -#define __builtin_expect(EXP, C) (EXP) -#else -#define UNUSED __attribute__((unused)) -#endif - -#define USE_P2O_MAPPER(op_name, class_name) \ - namespace paddle2onnx { \ - extern int Touch##op_name##class_name(); \ - int op_name##__registry__ UNUSED = Touch##op_name##class_name(); \ - } - -USE_P2O_MAPPER(quantize_linear, QuantizeLinearMapper) -USE_P2O_MAPPER(dequantize_linear, DequantizeLinearMapper) -USE_P2O_MAPPER(batch_norm, BatchNormMapper) -USE_P2O_MAPPER(dropout, DropoutMapper) -USE_P2O_MAPPER(layer_norm, LayerNormMapper) -USE_P2O_MAPPER(rnn, RnnMapper) -USE_P2O_MAPPER(softmax_with_cross_entropy, SoftmaxCrossEntropyLossMapper) -USE_P2O_MAPPER(affine_channel, AffineChannelMapper) -USE_P2O_MAPPER(conv3d, Conv3dMapper) -USE_P2O_MAPPER(conv2d_transpose, Conv2dTransposeMapper) -USE_P2O_MAPPER(depthwise_conv2d_transpose, Conv2dTransposeMapper) -USE_P2O_MAPPER(conv2d, Conv2dMapper) -USE_P2O_MAPPER(depthwise_conv2d, Conv2dMapper) -USE_P2O_MAPPER(pool3d, Pool3dMapper) -USE_P2O_MAPPER(max_pool3d_with_index, Pool3dMapper) -USE_P2O_MAPPER(instance_norm, InstanceNormMapper) -USE_P2O_MAPPER(pool2d, Pool2dMapper) -USE_P2O_MAPPER(max_pool2d_with_index, Pool2dMapper) -USE_P2O_MAPPER(shape, ShapeMapper) -USE_P2O_MAPPER(bilinear_interp, InterpolateMapper) -USE_P2O_MAPPER(bilinear_interp_v2, InterpolateMapper) -USE_P2O_MAPPER(nearest_interp_v2, InterpolateMapper) -USE_P2O_MAPPER(bicubic_interp_v2, InterpolateMapper) -USE_P2O_MAPPER(linear_interp_v2, InterpolateMapper) -USE_P2O_MAPPER(trilinear_interp_v2, InterpolateMapper) -USE_P2O_MAPPER(data_norm, DataNormMapper) -USE_P2O_MAPPER(norm, NormMapper) -USE_P2O_MAPPER(group_norm, GroupNormMapper) -USE_P2O_MAPPER(pad3d, Pad3DMapper) -USE_P2O_MAPPER(relu, ActivationMapper) -USE_P2O_MAPPER(relu6, Relu6Mapper) -USE_P2O_MAPPER(tanh, ActivationMapper) -USE_P2O_MAPPER(log, ActivationMapper) -USE_P2O_MAPPER(sigmoid, ActivationMapper) -USE_P2O_MAPPER(sqrt, ActivationMapper) -USE_P2O_MAPPER(softplus, ActivationMapper) -USE_P2O_MAPPER(exp, ActivationMapper) -USE_P2O_MAPPER(floor, ActivationMapper) -USE_P2O_MAPPER(cos, ActivationMapper) -USE_P2O_MAPPER(sin, ActivationMapper) -USE_P2O_MAPPER(round, ActivationMapper) -USE_P2O_MAPPER(abs, ActivationMapper) -USE_P2O_MAPPER(acos, ActivationMapper) -USE_P2O_MAPPER(asin, ActivationMapper) -USE_P2O_MAPPER(atan, ActivationMapper) -USE_P2O_MAPPER(sinh, ActivationMapper) -USE_P2O_MAPPER(tan, ActivationMapper) -USE_P2O_MAPPER(ceil, ActivationMapper) -USE_P2O_MAPPER(cosh, ActivationMapper) -USE_P2O_MAPPER(softsign, ActivationMapper) -USE_P2O_MAPPER(sign, ActivationMapper) -USE_P2O_MAPPER(erf, ActivationMapper) -USE_P2O_MAPPER(reciprocal, ActivationMapper) -USE_P2O_MAPPER(leaky_relu, LeakyReluMapper) -USE_P2O_MAPPER(gelu, GeluMapper) -USE_P2O_MAPPER(selu, SeluMapper) -USE_P2O_MAPPER(prelu, PReluMapper) -USE_P2O_MAPPER(hard_sigmoid, HardSigmoidMapper) -USE_P2O_MAPPER(swish, SwishMapper) -USE_P2O_MAPPER(hard_swish, HardSwishMapper) -USE_P2O_MAPPER(softmax, SoftMaxMapper) -USE_P2O_MAPPER(brelu, BReluMapper) -USE_P2O_MAPPER(elu, EluMapper) -USE_P2O_MAPPER(hard_shrink, HardShrinkMapper) -USE_P2O_MAPPER(softshrink, SoftShrinkMapper) -USE_P2O_MAPPER(mish, MishMapper) -USE_P2O_MAPPER(square, SquareMapper) -USE_P2O_MAPPER(size, SizeMapper) -USE_P2O_MAPPER(rsqrt, RsqrtMapper) -USE_P2O_MAPPER(logsigmoid, LogSigmoidMapper) -USE_P2O_MAPPER(log_softmax, LogSoftmaxMapper) -USE_P2O_MAPPER(tanh_shrink, TanhShrinkMapper) -USE_P2O_MAPPER(thresholded_relu, ThresholdedReluMapper) -USE_P2O_MAPPER(log1p, Log1PMapper) -USE_P2O_MAPPER(log2, Log2Mapper) -USE_P2O_MAPPER(log10, Log10Mapper) -USE_P2O_MAPPER(silu, SiluMapper) -USE_P2O_MAPPER(elementwise_add, ElementwiseMapper) -USE_P2O_MAPPER(elementwise_sub, ElementwiseMapper) -USE_P2O_MAPPER(elementwise_div, ElementwiseMapper) -USE_P2O_MAPPER(elementwise_mul, ElementwiseMapper) -USE_P2O_MAPPER(elementwise_min, ElementwiseMapper) -USE_P2O_MAPPER(elementwise_max, ElementwiseMapper) -USE_P2O_MAPPER(elementwise_pow, ElementwiseMapper) -USE_P2O_MAPPER(elementwise_mod, ElementWiseModMapper) -USE_P2O_MAPPER(elementwise_floordiv, ElementWiseFloordivMapper) -USE_P2O_MAPPER(flatten2, Flatten2Mapper) -USE_P2O_MAPPER(squeeze2, Squeeze2Mapper) -USE_P2O_MAPPER(clip, ClipMapper) -USE_P2O_MAPPER(pixel_shuffle, PixelShuffleMapper) -USE_P2O_MAPPER(one_hot_v2, OneHotV2Mapper) -USE_P2O_MAPPER(expand_v2, ExpandV2Mapper) -USE_P2O_MAPPER(cumsum, CumsumMapper) -USE_P2O_MAPPER(dot, DotMapper) -USE_P2O_MAPPER(less_than, LessThanMapper) -USE_P2O_MAPPER(reshape2, Reshape2Mapper) -USE_P2O_MAPPER(not_equal, NotEqualMapper) -USE_P2O_MAPPER(arg_min, ArgMinMapper) -USE_P2O_MAPPER(matmul_v2, MatmulV2Mapper) -USE_P2O_MAPPER(greater_equal, GreaterEqualMapper) -USE_P2O_MAPPER(scale, ScaleMapper) -USE_P2O_MAPPER(expand, ExpandMapper) -USE_P2O_MAPPER(dist, DistMapper) -USE_P2O_MAPPER(expand_as_v2, ExpandAsMapper) -USE_P2O_MAPPER(p_norm, PNormMapper) -USE_P2O_MAPPER(split, SplitMapper) -USE_P2O_MAPPER(mean, MeanMapper) -USE_P2O_MAPPER(set_value, SetValueMapper) -USE_P2O_MAPPER(transpose2, Transpose2Mapper) -USE_P2O_MAPPER(temporal_shift, TemporalShiftMapper) -USE_P2O_MAPPER(mul, MulMapper) -USE_P2O_MAPPER(flip, FlipMapper) -USE_P2O_MAPPER(bmm, BmmMapper) -USE_P2O_MAPPER(index_select, IndexSelectMapper) -USE_P2O_MAPPER(stack, StackMapper) -USE_P2O_MAPPER(gather_nd, GatherNdMapper) -USE_P2O_MAPPER(lookup_table, LookupTableMapper) -USE_P2O_MAPPER(lookup_table_v2, LookupTableMapper) -USE_P2O_MAPPER(meshgrid, MeshgridMapper) -USE_P2O_MAPPER(scatter, ScatterMapper) -USE_P2O_MAPPER(assign, AssignMapper) -USE_P2O_MAPPER(share_data, AssignMapper) -USE_P2O_MAPPER(scatter_nd_add, ScatterNdAddMapper) -USE_P2O_MAPPER(top_k_v2, TopKV2Mapper) -USE_P2O_MAPPER(range, RangeMapper) -USE_P2O_MAPPER(logical_not, LogicalNotMapper) -USE_P2O_MAPPER(linspace, LinspaceMapper) -USE_P2O_MAPPER(eye, EyeMapper) -USE_P2O_MAPPER(where, WhereMapper) -USE_P2O_MAPPER(unique, UniqueMapper) -USE_P2O_MAPPER(assign_value, AssignValueMapper) -USE_P2O_MAPPER(unstack, UnstackMapper) -USE_P2O_MAPPER(fill_any_like, FillLikeMapper) -USE_P2O_MAPPER(fill_zeros_like, FillLikeMapper) -USE_P2O_MAPPER(partial_sum, PartialOpsMapper) -USE_P2O_MAPPER(partial_concat, PartialOpsMapper) -USE_P2O_MAPPER(top_k, TopKMapper) -USE_P2O_MAPPER(grid_sampler, GridSamplerMapper) -USE_P2O_MAPPER(sum, AddNMapper) -USE_P2O_MAPPER(unsqueeze2, Unsqueeze2Mapper) -USE_P2O_MAPPER(arg_max, ArgMaxMapper) -USE_P2O_MAPPER(cast, CastMapper) -USE_P2O_MAPPER(matmul, MatmulMapper) -USE_P2O_MAPPER(gather, GatherMapper) -USE_P2O_MAPPER(fill_constant, FillConstantMapper) -USE_P2O_MAPPER(take_along_axis, TakeAlongAxisMapper) -USE_P2O_MAPPER(greater_than, GreaterThanMapper) -USE_P2O_MAPPER(argsort, ArgsortMapper) -USE_P2O_MAPPER(pow, PowMapper) -USE_P2O_MAPPER(tile, TileMapper) -USE_P2O_MAPPER(logical_and, LogicalOpMapper) -USE_P2O_MAPPER(logical_or, LogicalOpMapper) -USE_P2O_MAPPER(logical_xor, LogicalOpMapper) -USE_P2O_MAPPER(fill_constant_batch_size_like, FillConstantBatchSizeLikeMapper) -USE_P2O_MAPPER(slice, SliceMapper) -USE_P2O_MAPPER(strided_slice, SliceMapper) -USE_P2O_MAPPER(where_index, NonZeroMapper) -USE_P2O_MAPPER(reduce_mean, ReduceMapper) -USE_P2O_MAPPER(reduce_sum, ReduceMapper) -USE_P2O_MAPPER(reduce_min, ReduceMapper) -USE_P2O_MAPPER(reduce_max, ReduceMapper) -USE_P2O_MAPPER(reduce_prod, ReduceMapper) -USE_P2O_MAPPER(logsumexp, ReduceMapper) -USE_P2O_MAPPER(reduce_all, ReduceMapper) -USE_P2O_MAPPER(reduce_any, ReduceMapper) -USE_P2O_MAPPER(concat, ConcatMapper) -USE_P2O_MAPPER(less_equal, LessEqualMapper) -USE_P2O_MAPPER(gaussian_random, GaussianRandomMapper) -USE_P2O_MAPPER(flatten_contiguous_range, FlattenMapper) -USE_P2O_MAPPER(equal, EqualMapper) -USE_P2O_MAPPER(yolo_box, YoloBoxMapper) -USE_P2O_MAPPER(multiclass_nms3, NMSMapper) -USE_P2O_MAPPER(roi_align, RoiAlignMapper) -USE_P2O_MAPPER(index_sample, IndexSampleMapper) -USE_P2O_MAPPER(mv, MvMapper) -#endif // WITH_PADDLE2ONNX_STATIC_INTERNAL_AT_COMPILING -#endif // WITH_PADDLE2ONNX_STATIC_INTERNAL diff --git a/paddle2onnx/onnx_reader.cc b/paddle2onnx/onnx_reader.cc deleted file mode 100644 index 0cf1160b1d6..00000000000 --- a/paddle2onnx/onnx_reader.cc +++ /dev/null @@ -1,124 +0,0 @@ -#include -#include -#include -#include -#include -#include "paddle2onnx/converter.h" -#include "paddle2onnx/mapper/exporter.h" -#include "paddle2onnx/optimizer/paddle2onnx_optimizer.h" - -namespace paddle2onnx { - -int32_t GetDataTypeFromOnnx(int dtype) { - if (dtype == ONNX_NAMESPACE::TensorProto::FLOAT) { - return 0; - } else if (dtype == ONNX_NAMESPACE::TensorProto::DOUBLE) { - return 1; - } else if (dtype == ONNX_NAMESPACE::TensorProto::UINT8) { - return 2; - } else if (dtype == ONNX_NAMESPACE::TensorProto::INT8) { - return 3; - } else if (dtype == ONNX_NAMESPACE::TensorProto::INT32) { - return 4; - } else if (dtype == ONNX_NAMESPACE::TensorProto::INT64) { - return 5; - } else if (dtype == ONNX_NAMESPACE::TensorProto::FLOAT16) { - return 6; - } - Assert(false, "Only support float/double/uint8/int32/int64/float16 in OnnxReader."); - return -1; -} - -OnnxReader::OnnxReader(const char* model_buffer, int buffer_size) { - ONNX_NAMESPACE::ModelProto model; - std::string content(model_buffer, model_buffer + buffer_size); - model.ParseFromString(content); - - std::set initializer_names; - for (auto i = 0; i < model.graph().initializer_size(); ++i) { - initializer_names.insert(model.graph().initializer(i).name()); - } - - num_outputs = model.graph().output_size(); - Assert(num_outputs <= 100, - "The number of outputs is exceed 100, unexpected situation."); - - num_inputs = 0; - for (int i = 0; i < model.graph().input_size(); ++i) { - if (initializer_names.find(model.graph().input(i).name()) != - initializer_names.end()) { - continue; - } - num_inputs += 1; - Assert(num_inputs <= 100, - "The number of inputs is exceed 100, unexpected situation."); - - inputs[i].dtype = - GetDataTypeFromOnnx(model.graph().input(i).type().tensor_type().elem_type()); - std::strcpy(inputs[i].name, model.graph().input(i).name().c_str()); - auto& shape = model.graph().input(i).type().tensor_type().shape(); - int dim_size = shape.dim_size(); - inputs[i].rank = dim_size; - inputs[i].shape = new int64_t[dim_size]; - for (int j = 0; j < dim_size; ++j) { - inputs[i].shape[j] = static_cast(shape.dim(j).dim_value()); - if (inputs[i].shape[j] <= 0) { - inputs[i].shape[j] = -1; - } - } - } - - for (int i = 0; i < num_outputs; ++i) { - std::strcpy(outputs[i].name, model.graph().output(i).name().c_str()); - outputs[i].dtype = - GetDataTypeFromOnnx(model.graph().output(i).type().tensor_type().elem_type()); - auto& shape = model.graph().output(i).type().tensor_type().shape(); - int dim_size = shape.dim_size(); - outputs[i].rank = dim_size; - outputs[i].shape = new int64_t[dim_size]; - for (int j = 0; j < dim_size; ++j) { - outputs[i].shape[j] = static_cast(shape.dim(j).dim_value()); - if (outputs[i].shape[j] <= 0) { - outputs[i].shape[j] = -1; - } - } - } -} - -bool RemoveMultiClassNMS(const char* model_buffer, int buffer_size, - char** out_model, int* out_model_size) { - ONNX_NAMESPACE::ModelProto model; - std::string content(model_buffer, model_buffer + buffer_size); - model.ParseFromString(content); - auto* graph = model.mutable_graph(); - int nms_index = -1; - std::vector inputs; - for (int i = 0; i < graph->node_size(); ++i) { - if (graph->node(i).op_type() == "MultiClassNMS") { - nms_index = -1; - for (int j = 0; j < graph->node(i).input_size(); ++j) { - inputs.push_back(graph->node(i).input(j)); - } - break; - } - } - graph->clear_output(); - for (size_t i = 0; i < inputs.size(); ++i) { - auto output = graph->add_output(); - output->set_name(inputs[i]); - auto type_proto = output->mutable_type(); - auto tensor_type_proto = type_proto->mutable_tensor_type(); - tensor_type_proto->set_elem_type(ONNX_NAMESPACE::TensorProto::FLOAT); - auto shape = tensor_type_proto->mutable_shape(); - shape->add_dim()->set_dim_value(-1); - shape->add_dim()->set_dim_value(-1); - shape->add_dim()->set_dim_value(-1); - } - auto optimized_model = ONNX_NAMESPACE::optimization::OptimizeOnnxModel(model); - *out_model_size = optimized_model.ByteSizeLong(); - *out_model = new char[*out_model_size]; - optimized_model.SerializeToArray(*out_model, *out_model_size); - return true; -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/optimize.py b/paddle2onnx/optimize.py deleted file mode 100755 index a4bf35b6e1d..00000000000 --- a/paddle2onnx/optimize.py +++ /dev/null @@ -1,46 +0,0 @@ -# Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -from __future__ import absolute_import - -import argparse -import sys -from paddle2onnx.utils import logging - - -def parse_arguments(): - parser = argparse.ArgumentParser() - parser.add_argument( - '--input_model', - required=True, - help='The path of input onnx model file.') - parser.add_argument( - '--output_model', - required=True, - help='The file path to write optimized onnx model file.') - parser.add_argument( - '--input_shape_dict', - default="", - help="The shape infos of inputs, e.g --input_shape_dict=\"{'image': [1, 3, 608, 608], 'scale_factor': [1, 2]}\"" - ) - return parser.parse_args() - - -if __name__ == '__main__': - args = parse_arguments() - import paddle2onnx.paddle2onnx_cpp2py_export as c_p2o - shape_dict = {} - if args.input_shape_dict != "": - shape_dict = eval(args.input_shape_dict) - c_p2o.optimize(args.input_model, args.output_model, shape_dict) - logging.info("Model optmized, saved in {}.".format(args.output_model)) diff --git a/paddle2onnx/optimizer/convert_fp32_to_fp16.cc b/paddle2onnx/optimizer/convert_fp32_to_fp16.cc deleted file mode 100644 index 6fbd0967909..00000000000 --- a/paddle2onnx/optimizer/convert_fp32_to_fp16.cc +++ /dev/null @@ -1,840 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/optimizer/convert_fp32_to_fp16.h" - -#include "paddle2onnx/utils/utils.h" - -namespace paddle2onnx { - -void ConvertFp32ToFp16::ConvertValToFloat16(float val, uint16_t* x) { - // Conversion routine adapted from - // http://stackoverflow.com/questions/1659440/32-bit-to-16-bit-floating-point-conversion - Bits v, s; - v.f = val; - uint32_t sign = v.si & sigN; - v.si ^= sign; - sign >>= shiftSign; // logical shift - s.si = mulN; - s.si = s.f * v.f; // correct subnormals - v.si ^= (s.si ^ v.si) & -(minN > v.si); - v.si ^= (infN ^ v.si) & -((infN > v.si) & (v.si > maxN)); - v.si ^= (nanN ^ v.si) & -((nanN > v.si) & (v.si > infN)); - v.ui >>= shift; // logical shift - v.si ^= ((v.si - maxD) ^ v.si) & -(v.si > maxC); - v.si ^= ((v.si - minD) ^ v.si) & -(v.si > subC); - *x = v.ui | sign; -} - -void ConvertFp32ToFp16::SortNodes(ONNX_NAMESPACE::ModelProto* model) { - // return the topo sort of nodes; - // 1. Get i2o_mapper and constant_nodes, i2o_mapper means the node map to its - // all output nodes, constant_nodes save all constant nodes. - // 2. Nodes without output nodes are first saved to new_nodes, and then - // cyclically delete the records of the node in i2o_mapper items, and nodes - // whose output nodes are empty are also saved to new_nodes in turn. - // 3. Store constant nodes in new_nodes. - // 4. Reverse new_nodes, then assign to nodes. - auto graph = model->mutable_graph(); - - // means the node map to its all output nodes - std::map> i2o_mapper; - // constant_nodes save all constant nodes. - std::vector constant_nodes; - // name map to its node - std::map name2node_mapper; - for (int64_t i = 0; i < graph->node_size(); i++) { - auto node = graph->mutable_node(i); - if (node->op_type() == "Constant") { - constant_nodes.push_back(*node); - continue; - } - name2node_mapper[node->name()] = *node; - for (int64_t in_index = 0; in_index < node->input_size(); in_index++) { - std::string input = node->input(in_index); - for (int64_t j = 0; j < graph->node_size(); j++) { - if (i == j) { - continue; - } - auto input_node = graph->mutable_node(j); - if (input_node->op_type() == "Constant") { - continue; - } - for (int64_t out_index = 0; out_index < input_node->output_size(); - out_index++) { - // find the pre node - if (input == input_node->output(out_index)) { - // does not find other input node before - if (i2o_mapper.find(input_node->name()) == i2o_mapper.end()) { - i2o_mapper[input_node->name()] = {node->name()}; - } else { - auto iter = - std::find(i2o_mapper[input_node->name()].begin(), - i2o_mapper[input_node->name()].end(), node->name()); - // not been found before - if (iter == i2o_mapper[input_node->name()].end()) { - i2o_mapper[input_node->name()].push_back(node->name()); - } - } - } - } - } - } - } - - // Store topologically sorted nodes - std::vector new_nodes; - - for (int64_t i = 0; i < graph->node_size(); i++) { - auto node = graph->mutable_node(i); - auto node_name = node->name(); - if (node->op_type() == "Constant") { - continue; - } - // Store those nodes that have no output first. - if (i2o_mapper.find(node_name) == i2o_mapper.end()) { - new_nodes.push_back(*node); - } - } - - int64_t index = 0; - while (index < new_nodes.size()) { - auto current_node = new_nodes[index]; - std::string current_node_name = current_node.name(); - for (auto iter = i2o_mapper.begin(); iter != i2o_mapper.end(); iter++) { - std::string input_node_name = iter->first; - std::vector* output_nodes_name = &iter->second; - if (output_nodes_name->empty()) { - continue; - } - auto in_inter = std::find(output_nodes_name->begin(), - output_nodes_name->end(), current_node_name); - // if find the pre node, erase current node name in i2o_mapper - if (in_inter != output_nodes_name->end()) { - output_nodes_name->erase(in_inter); - } - // if find on node that have no output, store it - if (output_nodes_name->empty()) { - new_nodes.push_back(name2node_mapper[input_node_name]); - } - } - index++; - } - - // store all constant node finally - for (auto& node : constant_nodes) { - new_nodes.push_back(node); - } - - // reverse the sorted nodes - std::reverse(new_nodes.begin(), new_nodes.end()); - - Assert(model->mutable_graph()->node_size() == new_nodes.size(), - "The number of nodes after topological sorting is not equal to the " - "number before sorting"); - // copy all new_nodes to graph - for (int64_t i = 0; i < graph->node_size(); i++) { - auto node = graph->mutable_node(i); - node->CopyFrom(new_nodes[i]); - } -} - -std::string ConvertFp32ToFp16::GenName(const std::string& prefix) { - int64_t name_index = 0; - auto iter = name_index_mapper.find(prefix); - if (iter != name_index_mapper.end()) { - name_index = iter->second; - name_index_mapper[prefix]++; - } else { - name_index_mapper[prefix] = 1; - } - return prefix + std::to_string(name_index); -} - -ONNX_NAMESPACE::ValueInfoProto* ConvertFp32ToFp16::MakeValueInfoFromTensor( - const ONNX_NAMESPACE::TensorProto& tensor) { - ONNX_NAMESPACE::ValueInfoProto* value_info = - new ONNX_NAMESPACE::ValueInfoProto(); - value_info->set_name(tensor.name()); - auto type_proto = value_info->mutable_type(); - auto tensor_type_proto = type_proto->mutable_tensor_type(); - tensor_type_proto->set_elem_type(tensor.data_type()); // TODO - auto shape = tensor_type_proto->mutable_shape(); - for (auto i = 0; i < tensor.dims_size(); i++) { - auto dim = tensor.dims(i); - if (dim < 0) { - auto dynamic_dim_name = GenName("DynamicDimension"); - shape->add_dim()->set_dim_param(dynamic_dim_name); - } else { - shape->add_dim()->set_dim_value(dim); - } - } - return value_info; -} - -ONNX_NAMESPACE::NodeProto* ConvertFp32ToFp16::MakeCastNode( - const std::string& op_name, const std::vector& inputs, - const std::vector& outputs, int32_t to_dtype) { - ONNX_NAMESPACE::NodeProto* node = new ONNX_NAMESPACE::NodeProto(); - node->set_name(op_name); - node->set_op_type("Cast"); - for (size_t i = 0; i < inputs.size(); ++i) { - node->add_input(inputs[i]); - } - for (size_t i = 0; i < outputs.size(); ++i) { - node->add_output(outputs[i]); - } - auto attr = node->add_attribute(); - attr->set_name("to"); - attr->set_i(static_cast(to_dtype)); - attr->set_type(ONNX_NAMESPACE::AttributeProto::INT); - return node; -} - -bool ConvertFp32ToFp16::GetTensorValue( - const ONNX_NAMESPACE::TensorProto& tensor, std::vector* value) { - auto dtype = tensor.data_type(); - if (dtype != ONNX_NAMESPACE::TensorProto::FLOAT) { - return false; - } - std::vector shape; - for (int64_t i = 0; i < tensor.dims_size(); i++) { - shape.push_back(tensor.dims(i)); - } - int64_t nums = 1; - for (auto& i : shape) nums *= i; - value->resize(nums); - memcpy(value->data(), tensor.raw_data().data(), nums * sizeof(float)); - return value->size(); -} - -// When the value of a tensor is greater than 10000, it is reserved as FP32 and -// not converted. -bool ConvertFp32ToFp16::KeepNodeType(ONNX_NAMESPACE::NodeProto* node) { - auto KeepType = [=](const ONNX_NAMESPACE::TensorProto& tensor) { - std::vector fp32_val; - GetTensorValue(tensor, &fp32_val); - for (auto i = 0; i < fp32_val.size(); i++) { - if (fp32_val[i] > 10000) { - return true; - } - } - return false; - }; - - for (auto attr_index = 0; attr_index < node->attribute_size(); attr_index++) { - auto attr = node->attribute(attr_index); - if (attr.has_t() && KeepType(attr.t())) { - return true; - } - for (auto t_index = 0; t_index < attr.tensors_size(); t_index++) { - if (KeepType(attr.tensors(t_index))) { - return true; - } - } - } - return false; -} - -void ConvertFp32ToFp16::ConvertTensorFloatToFloat16( - ONNX_NAMESPACE::TensorProto* tensor) { - if (tensor->data_type() == ONNX_NAMESPACE::TensorProto::FLOAT) { - if (tensor->float_data_size()) { - Assert(false, "No implemented! Please raise an issue to us."); - } - if (tensor->has_raw_data()) { - std::vector fp32_val; - GetTensorValue(*tensor, &fp32_val); - if (fp32_val.empty()) { - return; - } - converted_attr++; - tensor->set_data_type(ONNX_NAMESPACE::TensorProto::FLOAT16); - - std::vector fp16_val(fp32_val.size(), 0); - - float pos_min_val = max_finite_val_; - float pos_max_val = min_positive_val_; - float neg_min_val = -1 * max_finite_val_; - float neg_max_val = -1 * min_positive_val_; - - for (auto i = 0; i < fp32_val.size(); i++) { - if (0 < fp32_val[i] && fp32_val[i] < min_positive_val_) { - if (fp32_val[i] < pos_min_val) { - pos_min_val = fp32_val[i]; - } - fp32_val[i] = min_positive_val_; - } else if (0 > fp32_val[i] && fp32_val[i] > -1 * min_positive_val_) { - if (fp32_val[i] > neg_min_val) { - neg_min_val = fp32_val[i]; - } - fp32_val[i] = -1 * min_positive_val_; - } else if (fp32_val[i] > max_finite_val_) { - if (fp32_val[i] > pos_max_val) { - pos_max_val = fp32_val[i]; - } - fp32_val[i] = max_finite_val_; - } else if (fp32_val[i] < -1 * max_finite_val_) { - if (fp32_val[i] < neg_max_val) { - neg_max_val = fp32_val[i]; - } - fp32_val[i] = -1 * max_finite_val_; - } - ConvertValToFloat16(fp32_val[i], &fp16_val[i]); - } - if (pos_min_val < max_finite_val_ - 1) { - P2OLogger() << "[Info] the float32 number: " << pos_min_val - << " will be truncated to: " << min_positive_val_ - << std::endl; - } - if (pos_max_val > min_positive_val_ + 1) { - P2OLogger() << "[Info] the float32 number: " << pos_max_val - << " will be truncated to: " << max_finite_val_ - << std::endl; - } - if (neg_min_val > -1 * max_finite_val_ + 1) { - P2OLogger() << "[Info] the float32 number: " << neg_min_val - << " will be truncated to: " << -1 * min_positive_val_ - << std::endl; - } - if (neg_max_val < -1 * min_positive_val_ - 1) { - P2OLogger() << "[Info] the float32 number: " << neg_max_val - << " will be truncated to: " << -1 * max_finite_val_ - << std::endl; - } - tensor->set_raw_data(std::string((const char*)(fp16_val.data()), - fp16_val.size() * sizeof(uint16_t))); - } - } -} - -// return if the next node of name is Cast and its attr type is dtype. -bool ConvertFp32ToFp16::CastedTo(const std::string& name, - ONNX_NAMESPACE::ModelProto& model, - int64_t dtype) { - auto graph = model.mutable_graph(); - std::vector next_nodes; - for (auto i = 0; i < graph->node_size(); i++) { - auto n = graph->mutable_node(i); - for (auto i_index = 0; i_index < n->input_size(); i_index++) { - std::string input = n->input(i_index); - if (name == input) { - next_nodes.push_back(n); - } - } - } - bool casted = false; - for (auto node : next_nodes) { - if (node->op_type() == "Cast") { - for (auto attr_index = 0; attr_index < node->attribute_size(); - attr_index++) { - if (node->attribute(attr_index).has_i() && - node->attribute(attr_index).i() == dtype) { - casted = true; - break; - } - } - } - } - return casted; -} - -// return if the pre node of name is Cast and its attr type is dtype. -bool ConvertFp32ToFp16::CastedFrom(const std::string& name, - ONNX_NAMESPACE::ModelProto& model, - int64_t dtype) { - auto graph = model.mutable_graph(); - std::vector pre_nodes; - for (auto i = 0; i < graph->node_size(); i++) { - auto n = graph->mutable_node(i); - for (auto o_index = 0; o_index < n->output_size(); o_index++) { - std::string output = n->output(o_index); - if (name == output) { - pre_nodes.push_back(n); - } - } - } - bool casted = false; - for (auto node : pre_nodes) { - if (node->op_type() == "Cast") { - for (auto attr_index = 0; attr_index < node->attribute_size(); - attr_index++) { - if (node->attribute(attr_index).has_i() && - node->attribute(attr_index).i() == dtype) { - casted = true; - break; - } - } - } - } - return casted; -} - -// return if the name is the input of DEFAULT_OP_BLOCK_LIST -bool ConvertFp32ToFp16::IsInputOfOpBlock(const std::string& name, - ONNX_NAMESPACE::ModelProto& model) { - auto graph = model.mutable_graph(); - for (auto i = 0; i < graph->node_size(); i++) { - auto n = graph->mutable_node(i); - if (std::find(op_block_list_.begin(), op_block_list_.end(), n->op_type()) == - op_block_list_.end()) { - continue; - } - - for (auto i_index = 0; i_index < n->input_size(); i_index++) { - std::string input = n->input(i_index); - if (name == input) { - return true; - } - } - } - return false; -} - -bool ConvertFp32ToFp16::IsOutputOfOpBlockAndFP32Out( - const std::string& name, ONNX_NAMESPACE::ModelProto& model) { - auto graph = model.mutable_graph(); - for (auto i = 0; i < graph->node_size(); i++) { - auto n = graph->mutable_node(i); - if (std::find(op_block_list_.begin(), op_block_list_.end(), n->op_type()) == - op_block_list_.end() && - std::find(fp32_output_op_list.begin(), fp32_output_op_list.end(), - n->op_type()) == fp32_output_op_list.end()) { - continue; - } - for (auto o_index = 0; o_index < n->output_size(); o_index++) { - std::string output = n->output(o_index); - if (name == output) { - return true; - } - } - } - return false; -} - -void ConvertFp32ToFp16::KeepIoType(ONNX_NAMESPACE::ModelProto* model) { - auto graph = model->mutable_graph(); - for (auto i = 0; i < graph->input_size(); i++) { - auto input = graph->input(i); - if (input.type().tensor_type().elem_type() == - ONNX_NAMESPACE::TensorProto::FLOAT) { - // if the pre node is cast, and it is cast to float16, we do not need add - // Cast OP any more - if (CastedTo(input.name(), *model, 10)) { - graph_io_to_skip.push_back(input.name()); - continue; - } - std::string output_name = "graph_input_cast_" + std::to_string(i); - name_mapping[input.name()] = output_name; - graph_io_to_skip.push_back(input.name()); - std::string node_name = "graph_input_cast" + std::to_string(i); - auto new_value_info = graph->add_value_info(); - new_value_info->CopyFrom(input); - new_value_info->set_name(output_name); - new_value_info->mutable_type()->mutable_tensor_type()->set_elem_type( - ONNX_NAMESPACE::TensorProto::FLOAT16); - auto new_node = - MakeCastNode(node_name, {input.name()}, {output_name}, 10); - *(graph->add_node()) = (*new_node); - value_info_list.push_back(new_value_info); - io_casts.push_back(node_name); - } - } - for (auto i = 0; i < graph->output_size(); i++) { - auto output = graph->output(i); - if (output.type().tensor_type().elem_type() == - ONNX_NAMESPACE::TensorProto::FLOAT) { - // if the next node is cast, and it is cast to float, we do not need add - // Cast OP any more - if (CastedFrom(output.name(), *model, 1)) { - graph_io_to_skip.push_back(output.name()); - continue; - } - std::string output_name = "graph_output_cast_" + std::to_string(i); - name_mapping[output.name()] = output_name; - graph_io_to_skip.push_back(output.name()); - std::string node_name = "graph_output_cast" + std::to_string(i); - auto new_value_info = graph->add_value_info(); - new_value_info->CopyFrom(output); - new_value_info->set_name(output_name); - new_value_info->mutable_type()->mutable_tensor_type()->set_elem_type( - ONNX_NAMESPACE::TensorProto::FLOAT16); - auto new_node = - MakeCastNode(node_name, {output_name}, {output.name()}, 1); - *(graph->add_node()) = (*new_node); - value_info_list.push_back(new_value_info); - io_casts.push_back(node_name); - } - } -} - -void ConvertFp32ToFp16::ConvertAttribute(ONNX_NAMESPACE::ModelProto* model) { - proto_node new_node(*model); - queue.push_back(new_node); - - while (queue.size()) { - next_level.clear(); - for (auto q : queue) { - // process model proto - if (q.node_type == "model" && model->has_graph()) { - proto_node new_node(model->mutable_graph()); - next_level.push_back(new_node); - } - // process graph proto - if (q.node_type == "graph") { - for (auto i = 0; i < q.graph->node_size(); i++) { - auto n = q.graph->mutable_node(i); - if (std::find(io_casts.begin(), io_casts.end(), n->name()) != - io_casts.end()) { - continue; - } - for (auto i_index = 0; i_index < n->input_size(); i_index++) { - std::string* input = n->mutable_input(i_index); - auto iter = name_mapping.find(*input); - if (iter != name_mapping.end()) { - *input = iter->second; - } - } - for (auto o_index = 0; o_index < n->output_size(); o_index++) { - std::string* output = n->mutable_output(o_index); - auto iter = name_mapping.find(*output); - if (iter != name_mapping.end()) { - *output = iter->second; - } - } - // If the op type is in op_block_list_ or fp32_output_op_list, - // or needs to be kept without conversion, then store the node in - // node_list, - // which is convenient for adding cast op in the front or back - if (KeepNodeType(n)) { - if (n->op_type() != "Constant" && - std::find(node_list.begin(), node_list.end(), n) == - node_list.end()) { - node_list.push_back(n); - } else { - for (auto index = 0; index < q.graph->node_size(); index++) { - auto keep_type_node = q.graph->mutable_node(index); - bool is_pre_node = - std::find(keep_type_node->input().begin(), - keep_type_node->input().end(), - n->output()[0]) != keep_type_node->input().end(); - if (is_pre_node && std::find(node_list.begin(), node_list.end(), - n) == node_list.end()) { - node_list.push_back(keep_type_node); - Assert( - n->op_type() == "Constant", - "The node type be Constant, but it is: " + n->op_type()); - keep_type_tensors.push_back(n->output()[0]); - } - } - } - } else if ((std::find(op_block_list_.begin(), op_block_list_.end(), - n->op_type()) != op_block_list_.end() || - std::find(fp32_output_op_list.begin(), - fp32_output_op_list.end(), - n->op_type()) != fp32_output_op_list.end()) && - std::find(node_list.begin(), node_list.end(), n) == - node_list.end()) { - node_list.push_back(n); - } else { - std::string op_name = n->name(); - if (n->op_type() == "Cast") { - for (auto attr_index = 0; attr_index < n->attribute_size(); - attr_index++) { - auto attr = n->mutable_attribute(attr_index); - if (attr->name() == "to" && attr->i() == 1) { - attr->set_i(10); - break; - } - } - } - for (auto attr_index = 0; attr_index < n->attribute_size(); - attr_index++) { - proto_node new_node(n->mutable_attribute(attr_index)); - next_level.push_back(new_node); - } - } - } - } - - // process attribute proto - if (q.node_type == "attribute") { - if (q.attr->has_g()) { - proto_node new_node(q.attr->mutable_g()); - next_level.push_back(new_node); - } - - for (auto g_index = 0; g_index < q.attr->graphs_size(); g_index++) { - proto_node new_node(q.attr->mutable_graphs(g_index)); - next_level.push_back(new_node); - } - if (q.attr->has_t()) { - ConvertTensorFloatToFloat16(q.attr->mutable_t()); - } - - for (auto t_index = 0; t_index < q.attr->tensors_size(); t_index++) { - ConvertTensorFloatToFloat16(q.attr->mutable_tensors(t_index)); - } - } - - // process graph proto - if (q.node_type == "graph") { - for (auto init_index = 0; init_index < q.graph->initializer_size(); - init_index++) { - auto init = q.graph->mutable_initializer(init_index); - if (init->data_type() == ONNX_NAMESPACE::TensorProto::FLOAT) { - ConvertTensorFloatToFloat16(init); - auto new_value = MakeValueInfoFromTensor(*init); - value_info_list.push_back(new_value); - } - } - for (auto i_index = 0; i_index < q.graph->input_size(); i_index++) { - auto input = q.graph->mutable_input(i_index); - bool skip = - std::find(graph_io_to_skip.begin(), graph_io_to_skip.end(), - input->name()) != graph_io_to_skip.end(); - if (!skip && input->type().tensor_type().elem_type() == - ONNX_NAMESPACE::TensorProto::FLOAT) { - input->mutable_type()->mutable_tensor_type()->set_elem_type( - ONNX_NAMESPACE::TensorProto::FLOAT16); - value_info_list.push_back(input); - } - } - for (auto o_index = 0; o_index < q.graph->output_size(); o_index++) { - auto output = q.graph->mutable_output(o_index); - bool skip = - std::find(graph_io_to_skip.begin(), graph_io_to_skip.end(), - output->name()) != graph_io_to_skip.end(); - if (!skip && output->type().tensor_type().elem_type() == - ONNX_NAMESPACE::TensorProto::FLOAT) { - output->mutable_type()->mutable_tensor_type()->set_elem_type( - ONNX_NAMESPACE::TensorProto::FLOAT16); - value_info_list.push_back(output); - } - } - for (auto value_index = 0; value_index < q.graph->value_info_size(); - value_index++) { - auto value = q.graph->mutable_value_info(value_index); - bool skip = - std::find(graph_io_to_skip.begin(), graph_io_to_skip.end(), - value->name()) != graph_io_to_skip.end(); - - // in Resize op, when the dims of input tensor is zero. - bool zero_shape_constant = false; - for (auto i = 0; i < q.graph->node_size(); i++) { - auto n = q.graph->mutable_node(i); - if (n->op_type() != "Constant" || n->output(0) != value->name()) { - continue; - } - for (auto attr_index = 0; attr_index < n->attribute_size(); - attr_index++) { - auto attr = n->mutable_attribute(attr_index); - if (attr->has_t() && attr->t().dims_size() == 1 && - attr->t().dims(0) == 0) { - zero_shape_constant = true; - } - } - if (zero_shape_constant) break; - } - - // if it is a tensor that should keep type float - bool keep_type_tensor = - std::find(keep_type_tensors.begin(), keep_type_tensors.end(), - value->name()) != keep_type_tensors.end(); - if (!zero_shape_constant && !skip && !keep_type_tensor && - value->type().tensor_type().elem_type() == - ONNX_NAMESPACE::TensorProto::FLOAT) { - value->mutable_type()->mutable_tensor_type()->set_elem_type( - ONNX_NAMESPACE::TensorProto::FLOAT16); - value_info_list.push_back(value); - } - } - } - } - queue.clear(); - queue = next_level; - } - // the model is a FP16 model - if (!converted_attr) { - return; - } - - auto graph = model->mutable_graph(); - for (auto node : node_list) { - // Handle the case of fp32_output OPs - // Add cast op for node output - if (std::find(fp32_output_op_list.begin(), fp32_output_op_list.end(), - node->op_type()) != fp32_output_op_list.end()) { - for (auto o_index = 0; o_index < node->output_size(); o_index++) { - std::string* output = node->mutable_output(o_index); - for (auto v_index = 0; v_index < graph->value_info().size(); - v_index++) { - auto value_info = graph->mutable_value_info(v_index); - if (value_info->name() == *output) { - if (value_info->type().tensor_type().elem_type() == - ONNX_NAMESPACE::TensorProto::FLOAT16 && - !CastedTo(*output, *model, 10)) { - std::string input_name = GenName(node->name() + "_output_cast_"); - std::string node_name = GenName(node->name() + "_output_cast"); - auto new_node = - MakeCastNode(node_name, {input_name}, {*output}, 10); - *(graph->add_node()) = (*new_node); - *(node->mutable_output(o_index)) = input_name; - } else { - break; - } - } - } - } - continue; - } - - // Handle the case of custom OPs - if (std::find(custom_ops_.begin(), custom_ops_.end(), node->op_type()) != - custom_ops_.end()) { - // add cast op for node input - for (auto i_index = 0; i_index < node->input_size(); i_index++) { - std::string* input = node->mutable_input(i_index); - for (auto v_index = 0; v_index < graph->value_info().size(); - v_index++) { - auto value_info = graph->mutable_value_info(v_index); - if (value_info->name() == *input) { - if (value_info->type().tensor_type().elem_type() == - ONNX_NAMESPACE::TensorProto::FLOAT16 && - !CastedFrom(*input, *model, 1)) { - std::string output_name = GenName(node->name() + "_input_cast_"); - std::string node_name = GenName(node->name() + "_input_cast"); - auto new_node = - MakeCastNode(node_name, {*input}, {output_name}, 1); - *(graph->add_node()) = (*new_node); - *(node->mutable_input(i_index)) = output_name; - } else { - break; - } - } - } - } - // add cast op for node output - for (auto o_index = 0; o_index < node->output_size(); o_index++) { - std::string* output = node->mutable_output(o_index); - for (auto v_index = 0; v_index < graph->value_info().size(); - v_index++) { - auto value_info = graph->mutable_value_info(v_index); - if (value_info->name() == *output) { - if (value_info->type().tensor_type().elem_type() == - ONNX_NAMESPACE::TensorProto::FLOAT16 && - !CastedTo(*output, *model, 10)) { - std::string input_name = GenName(node->name() + "_output_cast_"); - std::string node_name = GenName(node->name() + "_output_cast"); - auto new_node = - MakeCastNode(node_name, {input_name}, {*output}, 10); - *(graph->add_node()) = (*new_node); - *(node->mutable_output(o_index)) = input_name; - } else if (value_info->type().tensor_type().elem_type() == - ONNX_NAMESPACE::TensorProto::FLOAT16) { - value_info->mutable_type()->mutable_tensor_type()->set_elem_type( - ONNX_NAMESPACE::TensorProto::FLOAT); - } else { - break; - } - } - } - } - } else { - // Handle the case of DEFAULT_OP_BLOCK_LIST OPs - for (auto i_index = 0; i_index < node->input_size(); i_index++) { - std::string* input = node->mutable_input(i_index); - for (auto value_index = 0; value_index < value_info_list.size(); - value_index++) { - auto value_info = value_info_list[value_index]; - if (value_info->has_name() && *input == value_info->name() && - !CastedFrom(*input, *model, 1)) { - auto new_value_info = model->mutable_graph()->add_value_info(); - new_value_info->CopyFrom(*value_info); - std::string output_name = GenName(node->name() + "_input_cast_"); - new_value_info->set_name(output_name); - new_value_info->mutable_type() - ->mutable_tensor_type() - ->set_elem_type(ONNX_NAMESPACE::TensorProto::FLOAT); - std::string node_name = GenName(node->name() + "_input_cast"); - auto new_node = MakeCastNode(node_name, {*input}, {output_name}, 1); - *(graph->add_node()) = (*new_node); - *(node->mutable_input(i_index)) = output_name; - break; - } - } - } - for (auto o_index = 0; o_index < node->output_size(); o_index++) { - std::string* output = node->mutable_output(o_index); - for (auto value_index = 0; value_index < value_info_list.size(); - value_index++) { - if (*output == value_info_list[value_index]->name() && - !CastedTo(*output, *model, 10)) { - auto new_value_info = model->mutable_graph()->add_value_info(); - new_value_info->CopyFrom(*value_info_list[value_index]); - std::string input_name = GenName(node->name() + "_output_cast_"); - new_value_info->set_name(input_name); - new_value_info->mutable_type() - ->mutable_tensor_type() - ->set_elem_type(ONNX_NAMESPACE::TensorProto::FLOAT); - std::string node_name = GenName(node->name() + "_output_cast"); - auto new_node = - MakeCastNode(node_name, {input_name}, {*output}, 10); - *(graph->add_node()) = (*new_node); - *(node->mutable_output(o_index)) = input_name; - break; - } - } - } - } - } -} - -bool ConvertFp32ToFp16::IsFP16Model(const ONNX_NAMESPACE::ModelProto& model) { - for (auto node : model.graph().node()) { - if (node.op_type() == "Cast") { - auto name = node.name(); - if (name.find("_output_cast") != name.npos || - name.find("_input_cast") != name.npos || - name.find("graph_output_cast") != name.npos || - name.find("graph_input_cast") != name.npos) { - return true; - } - } - } - return false; -} - -void ConvertFp32ToFp16::Convert(ONNX_NAMESPACE::ModelProto* model) { - op_block_list_.insert(op_block_list_.end(), DEFAULT_OP_BLOCK_LIST.begin(), - DEFAULT_OP_BLOCK_LIST.end()); - if (custom_ops_.size()) { - op_block_list_.insert(op_block_list_.end(), custom_ops_.begin(), - custom_ops_.end()); - } - shape_inference::InferShapes(*model); - // 1 if it is a FP16 model, skip this - if (IsFP16Model(*model)) { - P2OLogger() << "[Info] The input ONNX Model is a FP16 model." << std::endl; - return; - } - // 2 keep IO types - KeepIoType(model); - // 3 ConvertAttribute - ConvertAttribute(model); - // 4 sortnodes - SortNodes(model); -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/optimizer/convert_fp32_to_fp16.h b/paddle2onnx/optimizer/convert_fp32_to_fp16.h deleted file mode 100644 index 95159f51cb9..00000000000 --- a/paddle2onnx/optimizer/convert_fp32_to_fp16.h +++ /dev/null @@ -1,240 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include - -#include -#include -#include - -#include "paddle2onnx/mapper/mapper.h" -#include "paddle2onnx/parser/parser.h" -namespace paddle2onnx { - -struct proto_node { - public: - std::string node_type; // model, graph, node, arribute - ONNX_NAMESPACE::ModelProto* model; - ONNX_NAMESPACE::GraphProto* graph; - ONNX_NAMESPACE::NodeProto* node; - ONNX_NAMESPACE::AttributeProto* attr; - - explicit proto_node(ONNX_NAMESPACE::ModelProto new_model) { - node_type = "model"; - model = &new_model; - } - - explicit proto_node(ONNX_NAMESPACE::ModelProto* new_model) { - node_type = "model"; - model = new_model; - } - - explicit proto_node(ONNX_NAMESPACE::GraphProto new_graph) { - node_type = "graph"; - graph = &new_graph; - } - - explicit proto_node(ONNX_NAMESPACE::GraphProto* new_graph) { - node_type = "graph"; - graph = new_graph; - } - - explicit proto_node(ONNX_NAMESPACE::NodeProto new_node) { - node_type = "node"; - node = &new_node; - } - - explicit proto_node(ONNX_NAMESPACE::NodeProto* new_node) { - node_type = "node"; - node = new_node; - } - - explicit proto_node(ONNX_NAMESPACE::AttributeProto new_attribute) { - node_type = "attribute"; - attr = &new_attribute; - } - - explicit proto_node(ONNX_NAMESPACE::AttributeProto* new_attribute) { - node_type = "attribute"; - attr = new_attribute; - } -}; - -struct ConvertFp32ToFp16 { - public: - ConvertFp32ToFp16(float min_positive_val = 1e-7, float max_finite_val = 1e4, - bool keep_io_types = false, - bool disable_shape_infer = false, - const std::vector& op_block_list = {}, - const std::vector& node_block_list = {}) { - min_positive_val_ = min_positive_val; - max_finite_val_ = max_finite_val; - keep_io_types_ = keep_io_types; - disable_shape_infer_ = disable_shape_infer; - op_block_list_ = op_block_list; - node_block_list_ = node_block_list; - } - - void Convert(ONNX_NAMESPACE::ModelProto* model); - - ONNX_NAMESPACE::NodeProto* MakeCastNode( - const std::string& op_name, const std::vector& inputs, - const std::vector& outputs, int32_t to_dtype); - - ONNX_NAMESPACE::ValueInfoProto* MakeValueInfoFromTensor( - const ONNX_NAMESPACE::TensorProto& tensor); - - void KeepIoType(ONNX_NAMESPACE::ModelProto* model); - - void ConvertAttribute(ONNX_NAMESPACE::ModelProto* model); - - void ConvertTensorFloatToFloat16(ONNX_NAMESPACE::TensorProto* tensor); - - // return if keep the type of node - bool KeepNodeType(ONNX_NAMESPACE::NodeProto* node); - - bool GetTensorValue(const ONNX_NAMESPACE::TensorProto& tensor, - std::vector* value); - - // topo sort - void SortNodes(ONNX_NAMESPACE::ModelProto* model); - - void ConvertValToFloat16(float val, uint16_t* x); - - // return if the next node of name is Cast and its attr type is dtype. - bool CastedTo(const std::string& name, ONNX_NAMESPACE::ModelProto& model, - int64_t dtype); - // return if the pre node of name is Cast and its attr type is dtype. - bool CastedFrom(const std::string& name, ONNX_NAMESPACE::ModelProto& model, - int64_t dtype); - // return if the name is the input of DEFAULT_OP_BLOCK_LIST - bool IsInputOfOpBlock(const std::string& name, - ONNX_NAMESPACE::ModelProto& model); - - // return if the name is the input of DEFAULT_OP_BLOCK_LIST and - // fp32_output_op_list - bool IsOutputOfOpBlockAndFP32Out(const std::string& name, - ONNX_NAMESPACE::ModelProto& model); - - void SetCustomOps(const std::map& custom_ops) { - if (custom_ops.size()) { - custom_ops_.clear(); - for (auto op : custom_ops) { - custom_ops_.push_back(op.second); - } - } - } - - void AddDisabledOpTypes(const std::vector& disable_fp16_ops) { - op_block_list_.insert(op_block_list_.end(), disable_fp16_ops.begin(), - disable_fp16_ops.end()); - } - // If the input ONNX model is a FP16 model, return True - bool IsFP16Model(const ONNX_NAMESPACE::ModelProto& model); - - private: - union Bits { - float f; - int32_t si; - uint32_t ui; - }; - static const int shift = 13; - static const int shiftSign = 16; - - static const int32_t infN = 0x7F800000; - static const int32_t maxN = 0x477FE000; // max flt16 as flt32 - static const int32_t minN = 0x38800000; // min flt16 normal as flt32 - static const int32_t sigN = 0x80000000; // sign bit - - static constexpr int32_t infC = infN >> shift; - static constexpr int32_t nanN = (infC + 1) - << shift; // minimum flt16 nan as float32 - static constexpr int32_t maxC = maxN >> shift; - static constexpr int32_t minC = minN >> shift; - static constexpr int32_t sigC = sigN >> shiftSign; - - static const int32_t mulN = 0x52000000; // (1 << 23) / minN - static const int32_t mulC = 0x33800000; // minN / (1 << (23 - shift)) - static const int32_t subC = 0x003FF; // max flt32 subnormal downshifted - static const int32_t norC = 0x00400; // min flt32 normal downshifted - - static constexpr int32_t maxD = infC - maxC - 1; - static constexpr int32_t minD = minC - subC - 1; - - float min_positive_val_ = 1e-7; - float max_finite_val_ = 1e4; - bool keep_io_types_ = false; - bool disable_shape_infer_ = false; - std::vector op_block_list_ = {}; - std::vector node_block_list_ = {}; - - std::vector custom_ops_ = {"AdaptivePool2d", "MultiClassNMS"}; - - int64_t converted_attr = 0; - - std::map name_mapping; - std::vector graph_io_to_skip; - std::vector value_info_list; - std::vector io_casts; - - std::vector node_list; - - std::vector queue; - std::vector next_level; - - std::map name_index_mapper; - // int64_t name_index = 0; - std::string GenName(const std::string& prefix); - - // save the tensor names that should keep data type - std::vector keep_type_tensors; - - // The input can be FP16, but the output can only be fp32 - std::vector fp32_output_op_list = {"RandomNormalLike"}; - - std::vector DEFAULT_OP_BLOCK_LIST = { - "ArrayFeatureExtractor", - "ReduceMean", // this op may cause wrong results on FP16 - "Binarizer", - "CastMap", - "CategoryMapper", - "DictVectorizer", - "FeatureVectorizer", - "Imputer", - "LabelEncoder", - "LinearClassifier", - "LinearRegressor", - "Normalizer", - "OneHotEncoder", - "RandomUniformLike", - "SVMClassifier", - "SVMRegressor", - "Scaler", - "TreeEnsembleClassifier", - "TreeEnsembleRegressor", - "ZipMap", - "NonMaxSuppression", - "TopK", - "RoiAlign", - "Resize", - "Range", - "CumSum", - "Min", - "Max", - "Upsample", // The following OP is added by Paddle developer - "EyeLike"}; -}; -} // namespace paddle2onnx diff --git a/paddle2onnx/optimizer/eliminate_non_transpose.h b/paddle2onnx/optimizer/eliminate_non_transpose.h deleted file mode 100644 index 3a1d28757fe..00000000000 --- a/paddle2onnx/optimizer/eliminate_non_transpose.h +++ /dev/null @@ -1,60 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/* - * SPDX-License-Identifier: Apache-2.0 - */ - -// ATTENTION: The code in this file is highly EXPERIMENTAL. -// Adventurous users should note that the APIs will probably change. - -#pragma once - -#include "onnxoptimizer/pass.h" - -namespace ONNX_NAMESPACE { -namespace optimization { - -struct EliminateNonTranspose final : public PredicateBasedPass { - explicit EliminateNonTranspose() - : PredicateBasedPass(PassType::Nop, PassEfficiency::Complete, - PassOptimizationType::Compute) {} - - std::string getPassName() const override { return "eliminate_non_transpose"; } - - bool patternMatchPredicate(Node* node) override { - return node->kind() == kTranspose; - } - bool runTransform(Node* node, Graph& graph, - NodeDestroyType& destroy_current) override { - if (node->hasAttribute(kperm)) { - auto perm = node->is(kperm); - for (size_t i = 0; i < perm.size(); ++i) { - if (perm[i] != i) { - return false; - } - } - } - const bool replacing_success = - tryReplacingAllUsesWith(node->output(), node->input()); - if (!replacing_success) { - return false; - } - destroy_current = NodeDestroyType::DestroyOne; - return true; - } -}; - -} // namespace optimization -} // namespace ONNX_NAMESPACE diff --git a/paddle2onnx/optimizer/fuse_constant_cast.h b/paddle2onnx/optimizer/fuse_constant_cast.h deleted file mode 100644 index 0fbaa4d5542..00000000000 --- a/paddle2onnx/optimizer/fuse_constant_cast.h +++ /dev/null @@ -1,67 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/* - * SPDX-License-Identifier: Apache-2.0 - */ - -#pragma once - -// Before: -// B = Reshape(Constant) -// After: -// B = Constant (Constant with new shape) - -#include - -#include "onnx/defs/tensor_util.h" -#include "onnxoptimizer/pass.h" - -namespace ONNX_NAMESPACE { -namespace optimization { - -struct FuseConstantCast final : public PredicateBasedPass { - explicit FuseConstantCast() - : PredicateBasedPass(PassType::Fuse, PassEfficiency::Complete, - PassOptimizationType::Compute) {} - std::string getPassName() const override { return "fuse_constant_cast"; } - - bool patternMatchPredicate(Node* node) override { - return node->kind() == kCast && - node->inputs()[0]->node()->kind() == kConstant; - } - bool runTransform(Node* n, Graph& graph, - NodeDestroyType& destroy_current) override { - destroy_current = NodeDestroyType::DestroyZero; - - if (n->inputs()[0]->uses().size() > 1) { - return false; - } - - Node* cast = n; - Node* constant = n->inputs()[0]->node(); - Tensor t = constant->t(kvalue); - auto dtype = cast->i(kto); - t.elem_type() = dtype; - constant->t_(kvalue, std::move(t)); - if (!tryReplacingAllUsesWith(cast->output(), cast->inputs()[0])) { - return false; - } - destroy_current = NodeDestroyType::DestroyOne; - return true; - } -}; - -} // namespace optimization -} // namespace ONNX_NAMESPACE diff --git a/paddle2onnx/optimizer/fuse_constant_reshape.h b/paddle2onnx/optimizer/fuse_constant_reshape.h deleted file mode 100644 index ba886f5c1c0..00000000000 --- a/paddle2onnx/optimizer/fuse_constant_reshape.h +++ /dev/null @@ -1,138 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/* - * SPDX-License-Identifier: Apache-2.0 - */ - -#pragma once - -// Before: -// B = Reshape(Constant) -// After: -// B = Constant (Constant with new shape) - -#include - -#include "onnx/defs/tensor_util.h" -#include "onnxoptimizer/pass.h" - -namespace ONNX_NAMESPACE { -namespace optimization { - -struct FuseConstantReshape final : public PredicateBasedPass { - explicit FuseConstantReshape() - : PredicateBasedPass(PassType::Fuse, PassEfficiency::Complete, - PassOptimizationType::Compute) {} - std::string getPassName() const override { return "fuse_constant_reshape"; } - - bool patternMatchPredicate(Node* node) override { - return node->kind() == kReshape && - node->inputs()[0]->node()->kind() == kConstant; - } - bool runTransform(Node* n, Graph& graph, - NodeDestroyType& destroy_current) override { - destroy_current = NodeDestroyType::DestroyZero; - - // check if Constant is only used by Reshape - if (n->inputs()[0]->uses().size() > 1) { - return false; - } - - Node* reshape = n; - Node* constant = n->inputs()[0]->node(); - - // Process 'reshape' data - std::vector shape; - if (reshape->hasAttribute(kshape)) { - // opset 5 and below - shape = reshape->is(kshape); - } else { - // opset 6 and above - first check if 'reshape' has 'shape' input - // constant - if (reshape->inputs()[1]->node()->kind() != kConstant) { - return false; - } - if (reshape->inputs()[1]->uses().size() > 1) { - return false; - } - Node* shape_const = reshape->inputs()[1]->node(); - Tensor t = shape_const->t(kvalue); - shape = ParseData(&t); - } - - int allow_zero = 0; - Symbol sym = Symbol("allowzero"); - if (reshape->hasAttribute(sym)) { - allow_zero = reshape->i(sym); - } - - Tensor t = constant->t(kvalue); - const auto& ori_size = t.sizes(); - - // process 0 in shape - if (allow_zero != 0) { - for (size_t i = 0; i < shape.size(); ++i) { - if (shape[i] == 0) { - // illegal situation - if (ori_size.size() <= i) { - return false; - } - shape[i] = ori_size[i]; - } - } - } - - // process -1 in shape - int count_of_unkown = 0; - int index_of_unkown = -1; - for (size_t i = 0; i < shape.size(); ++i) { - if (shape[i] == -1) { - count_of_unkown += 1; - index_of_unkown = i; - } - } - // illegal situtaion - if (count_of_unkown > 1) { - return false; - } - int64_t numel = std::accumulate(ori_size.begin(), ori_size.end(), 1, - std::multiplies()); - if (index_of_unkown >= 0) { - int64_t value_of_unkown = -1 * numel / - std::accumulate(shape.begin(), shape.end(), 1, - std::multiplies()); - shape[index_of_unkown] = value_of_unkown; - } - - t.sizes().clear(); - t.sizes().insert(t.sizes().begin(), shape.begin(), - shape.begin() + shape.size()); - constant->t_(kvalue, std::move(t)); - - // update constant node - constant->output()->setSizes(reshape->output()->sizes()); - constant->output()->setElemType(reshape->output()->elemType()); - const bool replacing_success = - tryReplacingAllUsesWith(reshape->output(), reshape->inputs()[0]); - if (!replacing_success) { - return false; - } - destroy_current = NodeDestroyType::DestroyOne; - return true; - } -}; - -} // namespace optimization -} // namespace ONNX_NAMESPACE diff --git a/paddle2onnx/optimizer/fuse_constant_unsqueeze.h b/paddle2onnx/optimizer/fuse_constant_unsqueeze.h deleted file mode 100644 index e4fa7cdf46f..00000000000 --- a/paddle2onnx/optimizer/fuse_constant_unsqueeze.h +++ /dev/null @@ -1,107 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/* - * SPDX-License-Identifier: Apache-2.0 - */ - -#pragma once - -// Before: -// B = Unsqueeze(Constant, axes) -// After: -// B = Constant (Constant with new shape) - -#include - -#include "onnx/defs/tensor_util.h" -#include "onnxoptimizer/pass.h" - -namespace ONNX_NAMESPACE { -namespace optimization { - -struct FuseConstantUnsqueeze final : public PredicateBasedPass { - explicit FuseConstantUnsqueeze() - : PredicateBasedPass(PassType::Fuse, PassEfficiency::Complete, - PassOptimizationType::Compute) {} - std::string getPassName() const override { return "fuse_constant_unsqueeze"; } - - bool patternMatchPredicate(Node* node) override { - return node->kind() == kUnsqueeze && - node->inputs()[0]->node()->kind() == kConstant; - } - bool runTransform(Node* n, Graph& graph, - NodeDestroyType& destroy_current) override { - destroy_current = NodeDestroyType::DestroyZero; - - // check if Constant is only used by Reshape - if (n->inputs()[0]->uses().size() > 1) { - return false; - } - - Node* unsqueeze = n; - Node* constant = n->inputs()[0]->node(); - - // Process 'axes' data - std::vector axes; - if (unsqueeze->hasAttribute(kaxes)) { - // opset 13 below - axes = unsqueeze->is(kaxes); - } else { - // opset 13 and above - first check if 'unsqueeze' has 'axes' input - // constant - if (unsqueeze->inputs()[1]->node()->kind() != kConstant) { - return false; - } - if (unsqueeze->inputs()[1]->uses().size() > 1) { - return false; - } - Node* axes_const = unsqueeze->inputs()[1]->node(); - Tensor t = axes_const->t(kvalue); - axes = ParseData(&t); - } - - Tensor t = constant->t(kvalue); - const auto& ori_size = t.sizes(); - for (size_t i = 0; i < axes.size(); ++i) { - if (axes[i] < 0) { - axes[i] = axes[i] + ori_size.size() + i + 1; - } - } - - std::vector new_size(ori_size.begin(), ori_size.end()); - for (size_t i = 0; i < axes.size(); ++i) { - new_size.insert(new_size.begin() + axes[i], 1); - } - - t.sizes().clear(); - t.sizes().insert(t.sizes().begin(), new_size.begin(), - new_size.begin() + new_size.size()); - constant->t_(kvalue, std::move(t)); - - // update constant node - constant->output()->setSizes(unsqueeze->output()->sizes()); - constant->output()->setElemType(unsqueeze->output()->elemType()); - const bool replacing_success = - tryReplacingAllUsesWith(unsqueeze->output(), unsqueeze->inputs()[0]); - if (!replacing_success) { - return false; - } - destroy_current = NodeDestroyType::DestroyOne; - return true; - } -}; - -} // namespace optimization -} // namespace ONNX_NAMESPACE diff --git a/paddle2onnx/optimizer/fuse_paddle_conv_bias.h b/paddle2onnx/optimizer/fuse_paddle_conv_bias.h deleted file mode 100644 index 89ed8883a1e..00000000000 --- a/paddle2onnx/optimizer/fuse_paddle_conv_bias.h +++ /dev/null @@ -1,96 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/* - * SPDX-License-Identifier: Apache-2.0 - */ - -#pragma once - -// Only support conv2d + bias now - -#include - -#include "onnx/defs/tensor_util.h" -#include "onnxoptimizer/pass.h" - -namespace ONNX_NAMESPACE { -namespace optimization { - -struct FusePaddleConvBias final : public PredicateBasedPass { - explicit FusePaddleConvBias() - : PredicateBasedPass(PassType::Fuse, PassEfficiency::Complete, - PassOptimizationType::Compute) {} - std::string getPassName() const override { return "fuse_paddle_conv_bias"; } - - bool patternMatchPredicate(Node* node) override { - return node->kind() == kAdd && node->inputs()[0]->node()->kind() == kConv && - node->inputs()[1]->node()->kind() == kConstant && - node->inputs()[0]->node()->inputs()[1]->node()->kind() == kConstant; - } - bool runTransform(Node* n, Graph& graph, - NodeDestroyType& destroy_current) override { - destroy_current = NodeDestroyType::DestroyZero; - - // check if Conv is only used by Add - if (n->inputs()[0]->uses().size() > 1) { - return false; - } - // check if bias is only used by Add - if (n->inputs()[1]->uses().size() > 1) { - return false; - } - - Node* add = n; - Node* conv = n->inputs()[0]->node(); - Node* bias = n->inputs()[1]->node(); - Node* weight = conv->inputs()[1]->node(); - - if (conv->inputs().size() > 2) { - return false; - } - - Tensor bias_tensor = bias->t(kvalue); - Tensor weight_tensor = weight->t(kvalue); - const auto& bias_shape = bias_tensor.sizes(); - const auto& weight_shape = weight_tensor.sizes(); - if (bias_shape.size() != 4 || bias_shape.size() != weight_shape.size()) { - return false; - } - if (bias_shape[0] != 1 || bias_shape[2] != 1 || bias_shape[3] != 1) { - return false; - } - if (bias_shape[1] != weight_shape[0]) { - return false; - } - // reshape bias node - bias_tensor.sizes().clear(); - bias_tensor.sizes().push_back(weight_shape[0]); - bias->t_(kvalue, std::move(bias_tensor)); - - conv->addInput(bias->outputs()[0]); - conv->output()->setSizes(add->output()->sizes()); - conv->output()->setElemType(add->output()->elemType()); - const bool replacing_success = - tryReplacingAllUsesWith(add->output(), add->inputs()[0]); - if (!replacing_success) { - return false; - } - destroy_current = NodeDestroyType::DestroyOne; - return true; - } -}; - -} // namespace optimization -} // namespace ONNX_NAMESPACE diff --git a/paddle2onnx/optimizer/fuse_squeeze_act_unsqueeze.h.bak b/paddle2onnx/optimizer/fuse_squeeze_act_unsqueeze.h.bak deleted file mode 100644 index dc5821b59ec..00000000000 --- a/paddle2onnx/optimizer/fuse_squeeze_act_unsqueeze.h.bak +++ /dev/null @@ -1,102 +0,0 @@ -/* - * SPDX-License-Identifier: Apache-2.0 - */ - -#pragma once - -// Before: -// axis = n -// A = Squeeze(x, axis) -// B = Act(A) -// C = Unsqueeze(x, axis) -// After: -// C = act(x) - -#include - -#include "onnx/defs/tensor_util.h" -#include "onnxoptimizer/pass.h" - -namespace ONNX_NAMESPACE { -namespace optimization { - -struct FuseSqueezeActUnsqueeze final : public PredicateBasedPass { - explicit FuseSqueezeActUnsqueeze() - : PredicateBasedPass(PassType::Fuse, PassEfficiency::Complete, - PassOptimizationType::Compute) {} - std::string getPassName() const override { return "fuse_squeeze_act_unsqueeze"; } - - bool patternMatchPredicate(Node* node) override { - // if you want more act supported - // just modify here - return node->kind() == kUnqueeze && - node->inputs()[0]->node()->kind() == kLeakyRelu && - node->inputs()[0]->node()->inputs()[0] == kSqueeze; - } - bool runTransform(Node* n, Graph& graph, - NodeDestroyType& destroy_current) override { - destroy_current = NodeDestroyType::DestroyZero; - - if (n->inputs()[0]->uses().size() > 1) { - return false; - } - if (n->inputs()[0]->node->inputs[0]->uses().size() > 1) { - return false; - } - - Node* unsqueeze = n; - Node* act = n->inputs()[0]->node(); - Node* squeeze = act->inputs()[0]->node(); - - // Process 'axes' data - std::vector unsqueeze_axes; - if (unsqueeze->hasAttribute(kaxes)) { - // opset 13 below - unsqueeze_axes = unsqueeze->is(kaxes); - } else { - // opset 13 and above - first check if 'unsqueeze' has 'axes' input - // constant - if (unsqueeze->inputs()[1]->node()->kind() != kConstant) { - return false; - } - if (unsqueeze->inputs()[1]->uses().size() > 1) { - return false; - } - Node* axes_const = unsqueeze->inputs()[1]->node(); - Tensor t = axes_const->t(kvalue); - unsqueeze_axes = ParseData(&t); - } - - Tensor t = constant->t(kvalue); - const auto& ori_size = t.sizes(); - for (size_t i = 0; i < axes.size(); ++i) { - if (axes[i] < 0) { - axes[i] = axes[i] + ori_size.size() + i + 1; - } - } - - std::vector new_size(ori_size.begin(), ori_size.end()); - for (size_t i = 0; i < axes.size(); ++i) { - new_size.insert(new_size.begin() + axes[i], 1); - } - - t.sizes().clear(); - t.sizes().insert(t.sizes().begin(), new_size.begin(), - new_size.begin() + new_size.size()); - constant->t_(kvalue, std::move(t)); - - // update constant node - constant->output()->setSizes(unsqueeze->output()->sizes()); - constant->output()->setElemType(unsqueeze->output()->elemType()); - const bool replacing_success = - tryReplacingAllUsesWith(unsqueeze->output(), unsqueeze->inputs()[0]); - if (!replacing_success) { - return false; - } - destroy_current = NodeDestroyType::DestroyOne; - return true; - } -}; - -} // namespace optimization -} // namespace ONNX_NAMESPACE diff --git a/paddle2onnx/optimizer/fuse_unsqueeze_conv2d_squeeze.h b/paddle2onnx/optimizer/fuse_unsqueeze_conv2d_squeeze.h deleted file mode 100644 index e97a15c985d..00000000000 --- a/paddle2onnx/optimizer/fuse_unsqueeze_conv2d_squeeze.h +++ /dev/null @@ -1,170 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/* - * SPDX-License-Identifier: Apache-2.0 - */ - -#pragma once - -// Before: -// X = Tensor(N, C, H) -// B = Unsqueeze(X, 2) -// C = Conv2d(B) -// D = Squeeze(C, 2) -// After: -// D = Conv1d(X) - -#include - -#include "onnx/defs/tensor_util.h" -#include "onnxoptimizer/pass.h" - -namespace ONNX_NAMESPACE { -namespace optimization { - -struct FuseUnsqueezeConv2dSqueeze final : public PredicateBasedPass { - explicit FuseUnsqueezeConv2dSqueeze() - : PredicateBasedPass(PassType::Fuse, PassEfficiency::Complete, - PassOptimizationType::Compute) {} - - std::string getPassName() const override { - return "fuse_unsqueeze_conv2d_squeeze"; - } - - bool patternMatchPredicate(Node* node) override { - return node->kind() == kSqueeze && - node->inputs()[0]->node()->kind() == kConv && - node->inputs()[0]->node()->inputs()[0]->node()->kind() == kUnsqueeze; - } - - bool runTransform(Node* n, Graph& graph, - NodeDestroyType& destroy_current) override { - Node* squeeze_node = n; - Node* conv_node = n->inputs()[0]->node(); - Node* unsqueeze_node = conv_node->inputs()[0]->node(); - if (squeeze_node->inputs()[0]->uses().size() > 1) { - return false; - } - if (conv_node->inputs()[0]->uses().size() > 1) { - return false; - } - - Node* weight_node = conv_node->inputs()[1]->node(); - if (weight_node->kind() != kConstant) { - return false; - } - Tensor weight = weight_node->t(kvalue); - if (weight.sizes().size() != 4) { - return false; - } - if (weight.sizes()[2] != 1) { - return false; - } - { - std::vector axes; - if (squeeze_node->hasAttribute(kaxes)) { - // opset 12 and below - axes = squeeze_node->is(kaxes); - } else { - // opset 13 and above - if (squeeze_node->inputs()[1]->node()->kind() != kConstant) { - return false; - } - if (squeeze_node->inputs()[1]->uses().size() > 1) { - return false; - } - Tensor t = squeeze_node->inputs()[1]->node()->t(kvalue); - axes = ParseData(&t); - } - if (axes.size() != 1 || axes[0] != 2) { - return false; - } - } - { - std::vector axes; - if (unsqueeze_node->hasAttribute(kaxes)) { - // opset 12 and below - axes = unsqueeze_node->is(kaxes); - } else { - // opset 13 and above - if (unsqueeze_node->inputs()[1]->node()->kind() != kConstant) { - return false; - } - if (unsqueeze_node->inputs()[1]->uses().size() > 1) { - return false; - } - Tensor t = unsqueeze_node->inputs()[1]->node()->t(kvalue); - axes = ParseData(&t); - } - if (axes.size() != 1 || axes[0] != 2) { - return false; - } - } - // update conv weight - weight.sizes().erase(weight.sizes().begin() + 2); - weight_node->t_(kvalue, std::move(weight)); - - if (conv_node->hasAttribute(kdilations)) { - std::vector dilations = conv_node->is(kdilations); - if (dilations.size() != 2 || dilations[0] != 1) { - return false; - } - dilations.erase(dilations.begin() + 0); - conv_node->is_(kdilations, std::move(dilations)); - } - if (conv_node->hasAttribute(kkernel_shape)) { - std::vector kernel_shape = conv_node->is(kkernel_shape); - if (kernel_shape.size() != 2 || kernel_shape[0] != 1) { - return false; - } - kernel_shape.erase(kernel_shape.begin() + 0); - conv_node->is_(kkernel_shape, std::move(kernel_shape)); - } - if (conv_node->hasAttribute(kpads)) { - std::vector pads = conv_node->is(kpads); - if (pads.size() != 4 || pads[0] != 0 || pads[2] != 0) { - return false; - } - pads.erase(pads.begin() + 2); - pads.erase(pads.begin() + 0); - conv_node->is_(kpads, std::move(pads)); - } - if (conv_node->hasAttribute(kstrides)) { - std::vector strides = conv_node->is(kstrides); - if (strides.size() != 2 || strides[0] != 1) { - return false; - } - strides.erase(strides.begin() + 0); - conv_node->is_(kstrides, std::move(strides)); - } - - conv_node->replaceInput(0, unsqueeze_node->inputs()[0]); - if (!tryReplacingAllUsesWith(unsqueeze_node->output(), - unsqueeze_node->inputs()[0])) { - return false; - } - if (!tryReplacingAllUsesWith(squeeze_node->output(), - squeeze_node->inputs()[0])) { - return false; - } - // unsqueeze_node->destroy(); - // squeeze_node->destroy(); - // destroy_current = NodeDestroyType::DestroyZero; - return true; - } -}; - -} // namespace optimization -} // namespace ONNX_NAMESPACE diff --git a/paddle2onnx/optimizer/paddle2onnx_optimizer.cc b/paddle2onnx/optimizer/paddle2onnx_optimizer.cc deleted file mode 100644 index 0900688b7fa..00000000000 --- a/paddle2onnx/optimizer/paddle2onnx_optimizer.cc +++ /dev/null @@ -1,240 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/optimizer/paddle2onnx_optimizer.h" -#include -#include -#include "onnxoptimizer/optimize.h" -#include "paddle2onnx/optimizer/eliminate_non_transpose.h" -#include "paddle2onnx/optimizer/fuse_constant_cast.h" -#include "paddle2onnx/optimizer/fuse_constant_reshape.h" -#include "paddle2onnx/optimizer/fuse_constant_unsqueeze.h" -#include "paddle2onnx/optimizer/fuse_paddle_conv_bias.h" -#include "paddle2onnx/optimizer/fuse_unsqueeze_conv2d_squeeze.h" -#include "paddle2onnx/optimizer/replace_add_to_identity.h" -#include "paddle2onnx/optimizer/replace_mul_to_identity.h" -#include "paddle2onnx/utils/utils.h" - -#include "paddle2onnx/converter.h" - -namespace ONNX_NAMESPACE { -namespace optimization { - -ONNX_NAMESPACE::ModelProto OptimizeOnnxModel( - const ONNX_NAMESPACE::ModelProto& model_proto) { - OptimizerOption option; - option.passes.clear(); - option.passes.push_back("eliminate_identity"); - option.passes.push_back("eliminate_deadend"); - - auto optimized_model_proto = - ONNX_NAMESPACE::optimization::Optimize(model_proto, option.passes); - - // reinfer shape for this onnx model - auto graph = optimized_model_proto.mutable_graph(); - // clear all the type info of outputs - auto output_size = graph->output_size(); - for (size_t i = 0; i < output_size; ++i) { - graph->mutable_output(i)->clear_type(); - } - - try { - shape_inference::InferShapes(optimized_model_proto); - } catch (const std::exception& e) { - P2OLogger(true) << "[ERROR] Failed to reinfer shape for this model." - << std::endl; - P2OLogger(true) << e.what() << std::endl; - } - return optimized_model_proto; -} - -std::shared_ptr LoadModelFromFile( - const std::string& file_path) { - auto model_proto = std::make_shared(); - std::ifstream fin(file_path, std::ios::in | std::ios::binary); - if (!fin.is_open()) { - P2OLogger(true) - << "Failed to read model file: " << file_path - << ", please make sure your model file or file path is valid." - << std::endl; - return model_proto; - } - std::string contents; - fin.seekg(0, std::ios::end); - contents.clear(); - contents.resize(fin.tellg()); - fin.seekg(0, std::ios::beg); - fin.read(&(contents.at(0)), contents.size()); - fin.close(); - - if (!model_proto->ParseFromString(contents)) { - P2OLogger(true) << "Failed to load ONNX model from file." << std::endl; - return model_proto; - } - return model_proto; -} - -bool OptimizePaddle2ONNX(const std::string& model_path, - const std::string& optimized_model_path, - const OptimizerOption& option) { - auto model_proto = LoadModelFromFile(model_path); - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - - auto optimized_model_proto = ONNX_NAMESPACE::optimization::Optimize( - *(model_proto.get()), option.passes); - std::string optimized_model_str; - if (!optimized_model_proto.SerializeToString(&optimized_model_str)) { - P2OLogger(true) << "Failed to serialize the optimized model protobuf." - << std::endl; - return false; - } - - std::fstream out(optimized_model_path, std::ios::out | std::ios::binary); - if (!out) { - P2OLogger(true) << "Failed to write the optimized model to disk at " - << optimized_model_path << "." << std::endl; - return false; - } - out << optimized_model_str; - out.close(); - return true; -} - -bool OptimizePaddle2ONNX( - const std::string& model_path, const std::string& optimized_model_path, - const std::map>& shape_infos, - const OptimizerOption& option) { - auto model_proto = LoadModelFromFile(model_path); - if (shape_infos.size() > 0) { - // reinfer shape for this onnx model - auto graph = model_proto->mutable_graph(); - // clear all the type info of outputs - auto output_size = graph->output_size(); - for (size_t i = 0; i < output_size; ++i) { - graph->mutable_output(i)->clear_type(); - } - // reset type info of inputs - auto input_size = graph->input_size(); - for (size_t i = 0; i < input_size; ++i) { - auto input_name = graph->input(i).name(); - auto iter = shape_infos.find(input_name); - if (iter != shape_infos.end()) { - auto tensor_type_proto = - graph->mutable_input(i)->mutable_type()->mutable_tensor_type(); - tensor_type_proto->clear_shape(); - auto shape = tensor_type_proto->mutable_shape(); - for (auto& dim : iter->second) { - shape->add_dim()->set_dim_value(dim); - } - } - } - - try { - shape_inference::InferShapes(*(model_proto.get())); - } catch (const std::exception& e) { - P2OLogger(true) << "[ERROR] Failed to reinfer shape for this model." - << std::endl; - P2OLogger(true) << e.what() << std::endl; - return false; - } - } - - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - ONNX_NAMESPACE::optimization::Optimizer::passes - .registerPass(); - - auto optimized_model_proto = ONNX_NAMESPACE::optimization::Optimize( - *(model_proto.get()), option.passes); - std::string optimized_model_str; - if (!optimized_model_proto.SerializeToString(&optimized_model_str)) { - P2OLogger(true) << "Failed to serialize the optimized model protobuf." - << std::endl; - return false; - } - - std::fstream out(optimized_model_path, std::ios::out | std::ios::binary); - if (!out) { - P2OLogger(true) << "Failed to write the optimized model to disk at " - << optimized_model_path << "." << std::endl; - return false; - } - out << optimized_model_str; - out.close(); - return true; -} - -bool Paddle2ONNXFP32ToFP16(const std::string& model_path, - const std::string& converted_model_path) { - std::ifstream fin(model_path, std::ios::in | std::ios::binary); - if (!fin.is_open()) { - P2OLogger(true) - << "Failed to read model file: " << model_path - << ", please make sure your model file or file path is valid." - << std::endl; - return false; - } - std::string contents; - fin.seekg(0, std::ios::end); - contents.clear(); - contents.resize(fin.tellg()); - fin.seekg(0, std::ios::beg); - fin.read(&(contents.at(0)), contents.size()); - fin.close(); - - char* out_model_ptr = nullptr; - int size = 0; - ConvertFP32ToFP16(contents.c_str(), contents.size(), &out_model_ptr, &size); - std::string onnx_proto(out_model_ptr, out_model_ptr + size); - - std::fstream out(converted_model_path, std::ios::out | std::ios::binary); - if (!out) { - P2OLogger(true) << "Failed to write the optimized model to disk at " - << converted_model_path << "." << std::endl; - return false; - } - out << onnx_proto; - out.close(); - return true; -} - -} // namespace optimization -} // namespace ONNX_NAMESPACE diff --git a/paddle2onnx/optimizer/paddle2onnx_optimizer.h b/paddle2onnx/optimizer/paddle2onnx_optimizer.h deleted file mode 100644 index 743291996a3..00000000000 --- a/paddle2onnx/optimizer/paddle2onnx_optimizer.h +++ /dev/null @@ -1,58 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include -#include -#include - -namespace ONNX_NAMESPACE { -namespace optimization { - -struct OptimizerOption { - std::vector passes; - OptimizerOption() { - passes.push_back("eliminate_identity"); - passes.push_back("eliminate_deadend"); - passes.push_back("fuse_constant_reshape"); - passes.push_back("fuse_constant_unsqueeze"); - passes.push_back("fuse_paddle_conv_bias"); - passes.push_back("fuse_consecutive_transposes"); - passes.push_back("eliminate_non_transpose"); - passes.push_back("replace_mul_to_identity"); - passes.push_back("replace_add_to_identity"); - passes.push_back("fuse_matmul_add_bias_into_gemm"); - passes.push_back("eliminate_identity"); - passes.push_back("eliminate_deadend"); - } -}; - -ONNX_NAMESPACE::ModelProto OptimizeOnnxModel( - const ONNX_NAMESPACE::ModelProto& model); - -bool OptimizePaddle2ONNX(const std::string& model_path, - const std::string& optimized_model_path, - const OptimizerOption& option = OptimizerOption()); - -bool OptimizePaddle2ONNX( - const std::string& model_path, const std::string& optimized_model_path, - const std::map>& shape_infos, - const OptimizerOption& option = OptimizerOption()); - -bool Paddle2ONNXFP32ToFP16(const std::string& model_path, - const std::string& optimized_model_path); - -} // namespace optimization -} // namespace ONNX_NAMESPACE diff --git a/paddle2onnx/optimizer/replace_add_to_identity.h b/paddle2onnx/optimizer/replace_add_to_identity.h deleted file mode 100644 index cfff7fcde22..00000000000 --- a/paddle2onnx/optimizer/replace_add_to_identity.h +++ /dev/null @@ -1,122 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/* - * SPDX-License-Identifier: Apache-2.0 - */ - -#pragma once - -// Before: -// X = Constant() all elements equal to 0, shape is () or (1,) -// Y = Tensor() -// C = X + Y -// After: -// C = Identity(Y) - -#include -#include -#include "onnx/defs/tensor_util.h" -#include "onnxoptimizer/pass.h" - -namespace ONNX_NAMESPACE { -namespace optimization { - -struct ReplaceAddToIdentity final : public PredicateBasedPass { - explicit ReplaceAddToIdentity() - : PredicateBasedPass(PassType::Fuse, PassEfficiency::Complete, - PassOptimizationType::Compute) {} - - std::string getPassName() const override { - return "replace_add_to_identity"; - } - - bool patternMatchPredicate(Node* node) override { - return node->kind() == kAdd && - (node->inputs()[0]->node()->kind() == kConstant || node->inputs()[1]->node()->kind() == kConstant); - } - - bool runTransform(Node* n, Graph& graph, - NodeDestroyType& destroy_current) override { - Node* add_node = n; - Node* add_ipt_0 = n->inputs()[0]->node(); - Node* add_ipt_1 = n->inputs()[1]->node(); - - if (add_ipt_0->kind() == kConstant) { - auto bias = add_ipt_0->t(kvalue); - if (bias.sizes().size() == 1 && bias.sizes()[0] != 1) { - return false; - } - if (bias.sizes().size() > 1) { - return false; - } - const auto& float_data = bias.floats(); - if (float_data.size() > 0 && fabs(float_data[0]) > 1e-05) { - return false; - } - const auto& double_data = bias.doubles(); - if (double_data.size() > 0 && fabs(double_data[0]) > 1e-05) { - return false; - } - const auto& int32_data = bias.int32s(); - if (int32_data.size() > 0 && int32_data[0] != 0) { - return false; - } - const auto& int64_data = bias.int64s(); - if (int64_data.size() > 0 && int64_data[0] != 0) { - return false; - } - if (float_data.size() == 0 && double_data.size() == 0 && int32_data.size() == 0 && int64_data.size() == 0) { - return false; - } - if (!tryReplacingAllUsesWith(add_node->output(), add_node->inputs()[1])) { - return false; - } - } else { - auto bias = add_ipt_1->t(kvalue); - if (bias.sizes().size() == 1 && bias.sizes()[0] != 1) { - return false; - } - if (bias.sizes().size() > 1) { - return false; - } - const auto& float_data = bias.floats(); - if (float_data.size() > 0 && fabs(float_data[0]) > 1e-05) { - return false; - } - const auto& double_data = bias.doubles(); - if (double_data.size() > 0 && fabs(double_data[0]) > 1e-05) { - return false; - } - const auto& int32_data = bias.int32s(); - if (int32_data.size() > 0 && int32_data[0] != 0) { - return false; - } - const auto& int64_data = bias.int64s(); - if (int64_data.size() > 0 && int64_data[0] != 0) { - return false; - } - if (float_data.size() == 0 && double_data.size() == 0 && int32_data.size() == 0 && int64_data.size() == 0) { - return false; - } - if (!tryReplacingAllUsesWith(add_node->output(), add_node->inputs()[0])) { - return false; - } - } - return true; - } -}; - -} // namespace optimization -} // namespace ONNX_NAMESPACE diff --git a/paddle2onnx/optimizer/replace_mul_to_identity.h b/paddle2onnx/optimizer/replace_mul_to_identity.h deleted file mode 100644 index 64516cbeb61..00000000000 --- a/paddle2onnx/optimizer/replace_mul_to_identity.h +++ /dev/null @@ -1,122 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/* - * SPDX-License-Identifier: Apache-2.0 - */ - -#pragma once - -// Before: -// X = Constant() all elements equal to 1, shape is () or (1,) -// Y = Tensor() -// C = X * Y -// After: -// C = Identity(Y) - -#include -#include -#include "onnx/defs/tensor_util.h" -#include "onnxoptimizer/pass.h" - -namespace ONNX_NAMESPACE { -namespace optimization { - -struct ReplaceMulToIdentity final : public PredicateBasedPass { - explicit ReplaceMulToIdentity() - : PredicateBasedPass(PassType::Fuse, PassEfficiency::Complete, - PassOptimizationType::Compute) {} - - std::string getPassName() const override { - return "replace_mul_to_identity"; - } - - bool patternMatchPredicate(Node* node) override { - return node->kind() == kMul && - (node->inputs()[0]->node()->kind() == kConstant || node->inputs()[1]->node()->kind() == kConstant); - } - - bool runTransform(Node* n, Graph& graph, - NodeDestroyType& destroy_current) override { - Node* mul_node = n; - Node* mul_ipt_0 = n->inputs()[0]->node(); - Node* mul_ipt_1 = n->inputs()[1]->node(); - - if (mul_ipt_0->kind() == kConstant) { - auto scale = mul_ipt_0->t(kvalue); - if (scale.sizes().size() == 1 && scale.sizes()[0] != 1) { - return false; - } - if (scale.sizes().size() > 1) { - return false; - } - const auto& float_data = scale.floats(); - if (float_data.size() > 0 && fabs(float_data[0] - 1.0) > 1e-05) { - return false; - } - const auto& double_data = scale.doubles(); - if (double_data.size() > 0 && fabs(double_data[0] - 1.0) > 1e-05) { - return false; - } - const auto& int32_data = scale.int32s(); - if (int32_data.size() > 0 && int32_data[0] != 1) { - return false; - } - const auto& int64_data = scale.int64s(); - if (int64_data.size() > 0 && int64_data[0] != 1) { - return false; - } - if (float_data.size() == 0 && double_data.size() == 0 && int32_data.size() == 0 && int64_data.size() == 0) { - return false; - } - if (!tryReplacingAllUsesWith(mul_node->output(), mul_node->inputs()[1])) { - return false; - } - } else { - auto scale = mul_ipt_1->t(kvalue); - if (scale.sizes().size() == 1 && scale.sizes()[0] != 1) { - return false; - } - if (scale.sizes().size() > 1) { - return false; - } - const auto& float_data = scale.floats(); - if (float_data.size() > 0 && fabs(float_data[0] - 1.0) > 1e-05) { - return false; - } - const auto& double_data = scale.doubles(); - if (double_data.size() > 0 && fabs(double_data[0] - 1.0) > 1e-05) { - return false; - } - const auto& int32_data = scale.int32s(); - if (int32_data.size() > 0 && int32_data[0] != 1) { - return false; - } - const auto& int64_data = scale.int64s(); - if (int64_data.size() > 0 && int64_data[0] != 1) { - return false; - } - if (float_data.size() == 0 && double_data.size() == 0 && int32_data.size() == 0 && int64_data.size() == 0) { - return false; - } - if (!tryReplacingAllUsesWith(mul_node->output(), mul_node->inputs()[0])) { - return false; - } - } - return true; - } -}; - -} // namespace optimization -} // namespace ONNX_NAMESPACE diff --git a/paddle2onnx/paddle_reader.cc b/paddle2onnx/paddle_reader.cc deleted file mode 100755 index 4d4aca513dc..00000000000 --- a/paddle2onnx/paddle_reader.cc +++ /dev/null @@ -1,78 +0,0 @@ -#include -#include -#include -#include -#include -#include "paddle2onnx/converter.h" -#include "paddle2onnx/mapper/exporter.h" -#include "paddle2onnx/parser/parser.h" - -namespace paddle2onnx { - -int32_t GetDataTypeFromPaddle(int dtype) { - if (dtype == P2ODataType::FP32) { - return 0; - } else if (dtype == P2ODataType::FP64) { - return 1; - } else if (dtype == P2ODataType::UINT8) { - return 2; - } else if (dtype == P2ODataType::INT8) { - return 3; - } else if (dtype == P2ODataType::INT32) { - return 4; - } else if (dtype == P2ODataType::INT64) { - return 5; - } - Assert(false, "Only support float/double/uint8/int32/int64 in PaddleReader."); - return -1; -} - -PaddleReader::PaddleReader(const char* model_buffer, int buffer_size) { - PaddleParser parser; - Assert(parser.Init(model_buffer, buffer_size), - "Failed to parse PaddlePaddle model."); - - num_inputs = parser.inputs.size(); - num_outputs = parser.outputs.size(); - for (int i = 0; i < num_inputs; ++i) { - std::strcpy(inputs[i].name, parser.inputs[i].name.c_str()); - inputs[i].rank = parser.inputs[i].Rank(); - inputs[i].shape = new int64_t[inputs[i].rank]; - for (int j = 0; j < inputs[i].rank; ++j) { - inputs[i].shape[j] = parser.inputs[i].shape[j]; - } - inputs[i].dtype = GetDataTypeFromPaddle(parser.inputs[i].dtype); - } - - for (int i = 0; i < num_outputs; ++i) { - std::strcpy(outputs[i].name, parser.outputs[i].name.c_str()); - outputs[i].rank = parser.outputs[i].Rank(); - outputs[i].shape = new int64_t[outputs[i].rank]; - for (int j = 0; j < outputs[i].rank; ++j) { - outputs[i].shape[j] = parser.outputs[i].shape[j]; - } - outputs[i].dtype = GetDataTypeFromPaddle(parser.outputs[i].dtype); - } - for (size_t i = 0; i < parser.NumOfOps(0); ++i) { - if (parser.GetOpDesc(0, i).type().find("quantize") != std::string::npos) { - is_quantize_model = true; - break; - } - } - for (size_t i = 0; i < parser.NumOfOps(0); ++i) { - if (parser.GetOpDesc(0, i).type().find("multiclass_nms3") != std::string::npos) { - has_nms = true; - auto& op = parser.GetOpDesc(0, i); - parser.GetOpAttr(op, "background_label", &nms_params.background_label); - parser.GetOpAttr(op, "keep_top_k", &nms_params.keep_top_k); - parser.GetOpAttr(op, "nms_eta", &nms_params.nms_eta); - parser.GetOpAttr(op, "nms_threshold", &nms_params.nms_threshold); - parser.GetOpAttr(op, "score_threshold", &nms_params.score_threshold); - parser.GetOpAttr(op, "nms_top_k", &nms_params.nms_top_k); - parser.GetOpAttr(op, "normalized", &nms_params.normalized); - break; - } - } -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/parser/parser.cc b/paddle2onnx/parser/parser.cc deleted file mode 100755 index d2c743a2d47..00000000000 --- a/paddle2onnx/parser/parser.cc +++ /dev/null @@ -1,862 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include "paddle2onnx/parser/parser.h" - -#include -#include -#include -#include - -#include "paddle2onnx/utils/utils.h" - -namespace paddle2onnx { -bool PaddleParser::LoadProgram(const std::string& model) { - prog = std::make_shared(); - std::ifstream fin(model, std::ios::in | std::ios::binary); - if (!fin.is_open()) { - P2OLogger() << "Failed to read model file: " << model - << ", please make sure your model file or file path is valid." - << std::endl; - return false; - } - - std::string contents; - fin.seekg(0, std::ios::end); - contents.clear(); - contents.resize(fin.tellg()); - fin.seekg(0, std::ios::beg); - fin.read(&(contents.at(0)), contents.size()); - fin.close(); - - if (!prog->ParseFromString(contents)) { - P2OLogger() << "Failed to parse paddlepaddle model from read content." - << std::endl; - return false; - } - return true; -} - -bool PaddleParser::LoadProgram(const void* model_buffer, int64_t model_size) { - prog = std::make_shared(); - if (!prog->ParseFromArray(model_buffer, model_size)) { - P2OLogger() << "Failed to parse PaddlePaddle model from memory buffer." - << std::endl; - return false; - } - return true; -} - -bool PaddleParser::GetParamNames(std::vector* var_names) { - var_names->clear(); - int block_size = prog->blocks_size(); - for (auto i = 0; i < block_size; ++i) { - auto block = prog->blocks(i); - int vars_size = block.vars_size(); - for (auto j = 0; j < vars_size; ++j) { - auto type = block.vars(j).type().type(); - if (type == framework::proto::VarType_Type::VarType_Type_SELECTED_ROWS) { - P2OLogger() - << "VarType of SELECTED_ROWS is not supported by Paddle2ONNX." - << std::endl; - return false; - } - if (type == framework::proto::VarType_Type::VarType_Type_FEED_MINIBATCH) { - continue; - } - if (type == paddle2onnx::framework::proto::VarType_Type:: - VarType_Type_FETCH_LIST) { - continue; - } - if (type == - paddle2onnx::framework::proto::VarType_Type::VarType_Type_READER) { - continue; - } - if (type == - paddle2onnx::framework::proto::VarType_Type::VarType_Type_RAW) { - continue; - } - if (!block.vars(j).persistable()) { - continue; - } - var_names->push_back(block.vars(j).name()); - } - } - std::sort(var_names->begin(), var_names->end()); - return true; -} - -bool PaddleParser::LoadParamsFromMemoryBuffer( - const std::string& params_buffer) { - params.clear(); - int64_t total_size = params_buffer.size(); - - std::vector var_names; - GetParamNames(&var_names); - - int read_size = 0; - while (read_size < total_size) { - auto index = params.size(); - if (index >= var_names.size()) { - P2OLogger() << "Unexcepted situation happend, while reading the " - "parameters of PaddlePaddle model." - << std::endl; - return false; - } - - { - // read version, we don't need this - uint32_t version; - params_buffer.copy(reinterpret_cast(&version), sizeof(version), - read_size); - read_size += sizeof(version); - } - { - // read lod_level, we don't use it - // this has to be zero, otherwise not support - uint64_t lod_level; - params_buffer.copy(reinterpret_cast(&lod_level), sizeof(lod_level), - read_size); - read_size += sizeof(lod_level); - if (lod_level != 0) { - P2OLogger() << "Only supports weight with lod_level = 0." << std::endl; - return false; - } - } - { - // Another version, we don't use it - uint32_t version; - params_buffer.copy(reinterpret_cast(&version), sizeof(version), - read_size); - read_size += sizeof(version); - } - { - // read size of TensorDesc - int32_t size; - params_buffer.copy(reinterpret_cast(&size), sizeof(size), - read_size); - read_size += sizeof(size); - // read TensorDesc - std::unique_ptr buf(new char[size]); - params_buffer.copy(reinterpret_cast(buf.get()), size, read_size); - read_size += size; - - std::unique_ptr - tensor_desc(new paddle2onnx::framework::proto::VarType_TensorDesc()); - tensor_desc->ParseFromArray(buf.get(), size); - - Weight weight; - - int32_t numel = 1; - int32_t data_type = tensor_desc->data_type(); - weight.dtype = data_type; - for (auto i = 0; i < tensor_desc->dims().size(); ++i) { - numel *= tensor_desc->dims()[i]; - weight.shape.push_back(tensor_desc->dims()[i]); - } - - // read weight data - weight.buffer.resize(numel * PaddleDataTypeSize(data_type)); - params_buffer.copy(weight.buffer.data(), - numel * PaddleDataTypeSize(data_type), read_size); - read_size += numel * PaddleDataTypeSize(data_type); - params[var_names[index]] = weight; - } - } - return true; -} - -bool PaddleParser::LoadParamsFromMemoryBuffer(const void* params_buffer, - int64_t params_size) { - params.clear(); - - const char* read_pointer = reinterpret_cast(params_buffer); - std::vector var_names; - GetParamNames(&var_names); - - while (params_size > 0) { - auto index = params.size(); - if (index >= var_names.size()) { - P2OLogger() << "Unexcepted situation happend, while reading the " - "parameters of PaddlePaddle model." - << std::endl; - return false; - } - - { - // read version, we don't need this - uint32_t version; - std::memcpy(&version, read_pointer, sizeof(version)); - read_pointer += sizeof(version); - params_size -= sizeof(version); - } - { - // read lod_level, we don't use it - // this has to be zero, otherwise not support - uint64_t lod_level; - std::memcpy(&lod_level, read_pointer, sizeof(lod_level)); - read_pointer += sizeof(lod_level); - params_size -= sizeof(lod_level); - if (lod_level != 0) { - P2OLogger() << "Only supports weight with lod_level = 0." << std::endl; - return false; - } - } - { - // Another version, we don't use it - uint32_t version; - std::memcpy(&version, read_pointer, sizeof(version)); - read_pointer += sizeof(version); - params_size -= sizeof(version); - } - { - // read size of TensorDesc - int32_t size; - std::memcpy(&size, read_pointer, sizeof(size)); - read_pointer += sizeof(size); - params_size -= sizeof(size); - // read TensorDesc - std::unique_ptr buf(new char[size]); - std::memcpy(buf.get(), read_pointer, size); - read_pointer += size; - params_size -= size; - - std::unique_ptr - tensor_desc(new paddle2onnx::framework::proto::VarType_TensorDesc()); - tensor_desc->ParseFromArray(buf.get(), size); - - Weight weight; - - int32_t numel = 1; - int32_t data_type = tensor_desc->data_type(); - weight.dtype = data_type; - for (auto i = 0; i < tensor_desc->dims().size(); ++i) { - numel *= tensor_desc->dims()[i]; - weight.shape.push_back(tensor_desc->dims()[i]); - } - - // read weight data - weight.buffer.resize(numel * PaddleDataTypeSize(data_type)); - std::memcpy(weight.buffer.data(), read_pointer, - numel * PaddleDataTypeSize(data_type)); - read_pointer += numel * PaddleDataTypeSize(data_type); - params_size -= numel * PaddleDataTypeSize(data_type); - params[var_names[index]] = weight; - } - } - return true; -} - -bool PaddleParser::LoadParams(const std::string& path) { - params.clear(); - std::ifstream is(path, std::ios::in | std::ios::binary); - if (!is.is_open()) { - P2OLogger() << "Cannot open file " << path << " to read." << std::endl; - return false; - } - is.seekg(0, std::ios::end); - int64_t total_size = is.tellg(); - is.seekg(0, std::ios::beg); - std::vector var_names; - GetParamNames(&var_names); - - int64_t read_size = 0; - while (read_size < total_size) { - { - // read version, we don't need this - uint32_t version; - read_size += sizeof(version); - is.read(reinterpret_cast(&version), sizeof(version)); - } - { - // read lod_level, we don't use it - // this has to be zero, otherwise not support - uint64_t lod_level; - read_size += sizeof(lod_level); - is.read(reinterpret_cast(&lod_level), sizeof(lod_level)); - Assert(lod_level == 0, - "Paddle2ONNX: Only support weight with lod_level = 0."); - } - { - // Another version, we don't use it - uint32_t version; - read_size += sizeof(version); - is.read(reinterpret_cast(&version), sizeof(version)); - } - { - // read size of TensorDesc - int32_t size; - read_size += sizeof(size); - is.read(reinterpret_cast(&size), sizeof(size)); - // read TensorDesc - std::unique_ptr buf(new char[size]); - read_size += size; - is.read(reinterpret_cast(buf.get()), size); - - std::unique_ptr - tensor_desc(new paddle2onnx::framework::proto::VarType_TensorDesc()); - tensor_desc->ParseFromArray(buf.get(), size); - - Weight weight; - - int32_t numel = 1; - int32_t data_type = tensor_desc->data_type(); - weight.dtype = data_type; - for (auto i = 0; i < tensor_desc->dims().size(); ++i) { - numel *= tensor_desc->dims()[i]; - weight.shape.push_back(tensor_desc->dims()[i]); - } - - // read weight data - weight.buffer.resize(numel * PaddleDataTypeSize(data_type)); - read_size += numel * PaddleDataTypeSize(data_type); - is.read(weight.buffer.data(), numel * PaddleDataTypeSize(data_type)); - auto index = params.size(); - if (index >= var_names.size()) { - P2OLogger() << "Unexcepted situation happend while reading parameters " - "of PaddlePaddle model." - << std::endl; - return false; - } - params[var_names[index]] = weight; - } - } - is.close(); - return true; -} - -int PaddleParser::NumOfBlocks() const { return prog->blocks_size(); } - -int PaddleParser::NumOfOps(int block_idx) const { - Assert(block_idx < NumOfBlocks(), - "block_idx is greater than number of blocks."); - return prog->blocks(block_idx).ops_size(); -} - -const framework::proto::OpDesc& PaddleParser::GetOpDesc(int32_t block_idx, - int32_t op_idx) const { - Assert(block_idx < NumOfBlocks(), - "block_idx is greater than number of blocks."); - Assert(op_idx < NumOfOps(block_idx), - "op_idx is greater than number of operators."); - return prog->blocks(block_idx).ops(op_idx); -} - -void PaddleParser::InitBlock() { - // if (ExistsDumplicateTensorName()) { - // return false; - // } - GetBlocksVarName2Id(); - GetBlocksOps(); - GetGlobalBlockInputOutputInfo(); -} - -bool PaddleParser::Init(const std::string& _model, const std::string& _params) { - std::vector weights; - if (!LoadProgram(_model)) { - P2OLogger() << "Failed to load program of PaddlePaddle model." << std::endl; - return false; - } - if (_params != "") { - if (!LoadParams(_params)) { - P2OLogger() << "Failed to load parameters of PaddlePaddle model." - << std::endl; - return false; - } - } - InitBlock(); - return true; -} - -bool PaddleParser::Init(const void* model_buffer, int64_t model_size, - const void* params_buffer, int64_t params_size) { - std::vector weights; - if (!LoadProgram(model_buffer, model_size)) { - P2OLogger() << "Failed to load program of PaddlePaddle model from memory." - << std::endl; - return false; - } - if (params_buffer != nullptr && params_size > 0) { - if (!LoadParamsFromMemoryBuffer(params_buffer, params_size)) { - P2OLogger() - << "Failed to load parameters of PaddlePaddle model from memory." - << std::endl; - return false; - } - } - InitBlock(); - return true; -} - -bool PaddleParser::IsConstantTensor(const int64_t& block_id, - const std::string& tensor_name) const { - Assert(block_id < _constant_ops.size(), - "block_id is out of range while calling IsConstantTensor."); - bool is_constant = false; - { - auto iter = _constant_ops[block_id].find(tensor_name); - is_constant = (iter != _constant_ops[block_id].end()); - } - if (!is_constant) { - auto iter = params.find(tensor_name); - is_constant = (iter != params.end()); - } - return is_constant; -} - -void PaddleParser::GetBlocksVarName2Id() { - _blocks_var_name2id.clear(); - _blocks_var_name2id.resize(prog->blocks_size()); - for (auto i = 0; i < prog->blocks_size(); ++i) { - for (auto j = 0; j < prog->blocks(i).vars_size(); ++j) { - _blocks_var_name2id[i][prog->blocks(i).vars(j).name()] = j; - } - } -} - -void PaddleParser::GetBlocksOps() { - is_quantized_model = false; - _blocks_ops.clear(); - _constant_ops.clear(); - _blocks_ops.resize(prog->blocks_size()); - _constant_ops.resize(prog->blocks_size()); - for (auto i = 0; i < prog->blocks_size(); ++i) { - _blocks_ops[i].reserve(prog->blocks(i).ops_size()); - for (auto j = 0; j < prog->blocks(i).ops_size(); ++j) { - _blocks_ops[i].push_back(&prog->blocks(i).ops(j)); - if (prog->blocks(i).ops(j).type() == "assign_value") { - _constant_ops[i][prog->blocks(i).ops(j).outputs(0).arguments(0)] = j; - } - // Determine whether the model is a quantized model, if it is a quantized - // model, set is_quantized_model to be true - if (!is_quantized_model && - prog->blocks(i).ops(j).type().find("quantize") != std::string::npos) { - is_quantized_model = true; - P2OLogger() << "[Info] The Paddle model is a quantized model. " - << std::endl; - } - } - } -} - -TensorInfo PaddleParser::GetTensorInfo( - const std::string& name, - const paddle2onnx::framework::proto::BlockDesc& block) const { - auto block_idx = block.idx(); - auto iter = _blocks_var_name2id[block_idx].find(name); - if (iter == _blocks_var_name2id[block_idx].end()) { - if (block_idx == 0) { - Assert(false, - "Cannot find " + name + " in _blocks_var_name2id(global block)."); - } else { - block_idx = block.parent_idx(); - iter = _blocks_var_name2id[block_idx].find(name); - Assert(iter != _blocks_var_name2id[block_idx].end(), - "Cannot find " + name + " in _blocks_var_name2id(parent block)."); - } - } - auto var_idx = iter->second; - - // Dangerous conversion, lod tensor array is under limited supporting - // Only works in some control flow situation - if (prog->blocks(block_idx).vars(var_idx).type().has_tensor_array()) { - auto tensor_array = - prog->blocks(block_idx).vars(var_idx).type().tensor_array(); - TensorInfo info; - info.is_tensor_array = true; - info.name = name; - info.dtype = tensor_array.tensor().data_type(); - for (auto i = 0; i < tensor_array.tensor().dims_size(); ++i) { - info.shape.push_back(tensor_array.tensor().dims(i)); - } - return info; - } - - auto tensor = prog->blocks(block_idx).vars(var_idx).type().lod_tensor(); - TensorInfo info; - info.name = name; - info.dtype = tensor.tensor().data_type(); - for (auto i = 0; i < tensor.tensor().dims_size(); ++i) { - info.shape.push_back(tensor.tensor().dims(i)); - } - - return info; -} - -bool PaddleParser::OpHasInput(int64_t block_id, int64_t op_id, - const std::string& name) const { - auto& block = prog->blocks(block_id); - auto& op = block.ops(op_id); - for (auto i = 0; i < op.inputs_size(); ++i) { - if (op.inputs(i).parameter() == name) { - if (op.inputs(i).arguments_size() > 0) { - return true; - } - } - } - return false; -} - -std::vector PaddleParser::GetOpInput( - int64_t block_id, int64_t op_id, const std::string& name) const { - auto& block = prog->blocks(block_id); - auto& op = block.ops(op_id); - std::vector inputs; - bool found = false; - for (auto i = 0; i < op.inputs_size(); ++i) { - if (op.inputs(i).parameter() == name) { - for (auto j = 0; j < op.inputs(i).arguments_size(); ++j) { - inputs.push_back(GetTensorInfo(op.inputs(i).arguments(j), block)); - found = true; - } - break; - } - } - Assert(found, "Cannot find input: " + name + " in operator: " + op.type()); - return inputs; -} - -bool PaddleParser::OpHasOutput(int64_t block_id, int64_t op_id, - const std::string& name) const { - auto& block = prog->blocks(block_id); - auto& op = block.ops(op_id); - for (auto i = 0; i < op.outputs_size(); ++i) { - if (op.outputs(i).parameter() == name) { - if (op.outputs(i).arguments_size() > 0) { - return true; - } - } - } - return false; -} - -std::vector PaddleParser::GetOpOutput( - int64_t block_id, int64_t op_id, const std::string& name) const { - auto& block = prog->blocks(block_id); - auto& op = block.ops(op_id); - std::vector outputs; - bool found = false; - for (auto i = 0; i < op.outputs_size(); ++i) { - if (op.outputs(i).parameter() == name) { - for (auto j = 0; j < op.outputs(i).arguments_size(); ++j) { - outputs.push_back(GetTensorInfo(op.outputs(i).arguments(j), block)); - found = true; - } - break; - } - } - Assert(found, "Cannot find output: " + name + " in operator: " + op.type()); - return outputs; -} - -bool PaddleParser::OpIsAttrVar(int64_t block_id, int64_t op_id, - const std::string& name) const { - bool is_attr_var = false; - auto& op = GetOpDesc(block_id, op_id); - for (auto i = 0; i < op.attrs_size(); ++i) { - if (op.attrs(i).name() == name && IsAttrVar(op, i)) { - is_attr_var = true; - break; - } - } - return is_attr_var; -} - -std::vector PaddleParser::GetOpAttrVar( - int64_t block_id, int64_t op_id, const std::string& name) const { - auto& block = prog->blocks(block_id); - auto& op = block.ops(op_id); - - bool found = false; - std::vector inputs; - for (auto i = 0; i < op.attrs_size(); ++i) { - if (op.attrs(i).name() == name) { - Assert(IsAttrVar(op, i), "Required AttrVar: " + name + - " type is Variable in operator: " + - op.type()); - // Case 1: Attribute is a single Var - if (op.attrs(i).has_var_name()) { - inputs.push_back(GetTensorInfo(op.attrs(i).var_name(), block)); - } else { // Case 2: Attribute is a List[Var] - for (int idx = 0; idx < op.attrs(i).vars_name_size(); ++idx) { - auto& var_name = op.attrs(i).vars_name(idx); - inputs.push_back(GetTensorInfo(var_name, block)); - } - } - found = true; - break; - } - } - Assert(found, "Cannot find AttrVar: " + name + " in operator: " + op.type()); - return inputs; -} - -bool PaddleParser::IsAttrVar(const paddle2onnx::framework::proto::OpDesc& op, - const int64_t& attr_id) const { - return op.attrs(attr_id).has_var_name() || - op.attrs(attr_id).vars_name_size() > 0; -} - -bool PaddleParser::OpHasAttr(const paddle2onnx::framework::proto::OpDesc& op, - const std::string& name) const { - bool found = false; - for (auto i = 0; i < op.attrs_size(); ++i) { - // set found to true when name is in op attrs and can use GetOpAttr to get - // value - if (op.attrs(i).name() == name) { - found = true; - break; - } - } - return found; -} - -void PaddleParser::GetOpAttr(const paddle2onnx::framework::proto::OpDesc& op, - const std::string& name, int64_t* res) const { - bool found = false; - for (auto i = 0; i < op.attrs_size(); ++i) { - if (op.attrs(i).name() == name) { - found = true; - if (IsAttrVar(op, i)) break; - Assert(op.attrs(i).has_i() || op.attrs(i).has_l(), - "Cannot find int32/int64 data from attr: " + name + " in op:" + - op.type()); - if (op.attrs(i).has_i()) { - *res = (int64_t)(op.attrs(i).i()); - } else { - *res = op.attrs(i).l(); - } - break; - } - } - Assert(found, "Cannot found attribute " + name + " in op: " + op.type()); -} - -void PaddleParser::GetOpAttr(const paddle2onnx::framework::proto::OpDesc& op, - const std::string& name, float* res) const { - bool found = false; - for (auto i = 0; i < op.attrs_size(); ++i) { - if (op.attrs(i).name() == name) { - found = true; - if (IsAttrVar(op, i)) break; - Assert(op.attrs(i).has_f(), "Cannot find float data from attr: " + name + - " in op: " + op.type()); - *res = op.attrs(i).f(); - break; - } - } - Assert(found, "Cannot found attribute " + name + " in op: " + op.type()); -} - -void PaddleParser::GetOpAttr(const paddle2onnx::framework::proto::OpDesc& op, - const std::string& name, bool* res) const { - bool found = false; - for (auto i = 0; i < op.attrs_size(); ++i) { - if (op.attrs(i).name() == name) { - found = true; - if (IsAttrVar(op, i)) break; - Assert(op.attrs(i).has_b(), "Cannot find bool data from attr: " + name + - " in op: " + op.type()); - *res = op.attrs(i).b(); - break; - } - } - Assert(found, "Cannot found attribute " + name + " in op: " + op.type()); -} - -void PaddleParser::GetOpAttr(const paddle2onnx::framework::proto::OpDesc& op, - const std::string& name, std::string* res) const { - bool found = false; - for (auto i = 0; i < op.attrs_size(); ++i) { - if (op.attrs(i).name() == name) { - found = true; - if (IsAttrVar(op, i)) break; - Assert(op.attrs(i).has_s(), "Cannot find string data from attr: " + name + - " in op: " + op.type()); - *res = op.attrs(i).s(); - break; - } - } - Assert(found, "Cannot found attribute " + name + " in op: " + op.type()); -} - -void PaddleParser::GetOpAttr(const paddle2onnx::framework::proto::OpDesc& op, - const std::string& name, - std::vector* res) const { - bool found = false; - res->clear(); - for (auto i = 0; i < op.attrs_size(); ++i) { - if (op.attrs(i).name() == name) { - found = true; - if (IsAttrVar(op, i)) break; - Assert(op.attrs(i).ints_size() >= 0 || op.attrs(i).longs_size() >= 0, - "Cannot find list of int32/int64 data from attr: " + name + - " in op: " + op.type()); - if (op.attrs(i).ints_size() > 0) { - for (auto j = 0; j < op.attrs(i).ints_size(); ++j) { - res->push_back(static_cast(op.attrs(i).ints(j))); - } - } else { - for (auto j = 0; j < op.attrs(i).longs_size(); ++j) { - res->push_back(op.attrs(i).longs(j)); - } - } - break; - } - } - Assert(found, "Cannot found attribute " + name + " in op: " + op.type()); -} - -void PaddleParser::GetOpAttr(const paddle2onnx::framework::proto::OpDesc& op, - const std::string& name, - std::vector* res) const { - bool found = false; - res->clear(); - for (auto i = 0; i < op.attrs_size(); ++i) { - if (op.attrs(i).name() == name) { - found = true; - if (IsAttrVar(op, i)) break; - Assert(op.attrs(i).floats_size() >= 0, - "Cannot find list of float data from attr: " + name + " in op: " + - op.type()); - for (auto j = 0; j < op.attrs(i).floats_size(); ++j) { - res->push_back(static_cast(op.attrs(i).floats(j))); - } - break; - } - } - Assert(found, "Cannot found attribute " + name + " in op: " + op.type()); -} - -void PaddleParser::GetOpAttr(const paddle2onnx::framework::proto::OpDesc& op, - const std::string& name, - std::vector* res) const { - bool found = false; - res->clear(); - for (auto i = 0; i < op.attrs_size(); ++i) { - if (op.attrs(i).name() == name) { - found = true; - if (IsAttrVar(op, i)) break; - Assert(op.attrs(i).float64s_size() >= 0, - "Cannot find list of double data from attr: " + name + " in op: " + - op.type()); - for (auto j = 0; j < op.attrs(i).float64s_size(); ++j) { - res->push_back(static_cast(op.attrs(i).float64s(j))); - } - break; - } - } - Assert(found, "Cannot found attribute " + name + " in op: " + op.type()); -} - -void PaddleParser::GetGlobalBlockInputOutputInfo() { - inputs.clear(); - outputs.clear(); - // record the origin order of Paddle model - std::vector inputs_with_no_order; - std::vector outputs_with_no_order; - std::vector input_order; - std::vector output_order; - - for (auto i = 0; i < prog->blocks(0).ops_size(); ++i) { - if (prog->blocks(0).ops(i).type() == "fetch") { - std::string name = prog->blocks(0).ops(i).inputs(0).arguments(0); - outputs_with_no_order.push_back(GetTensorInfo(name, prog->blocks(0))); - int64_t order = -1; - GetOpAttr(prog->blocks(0).ops(i), "col", &order); - output_order.push_back(order); - } else if (prog->blocks(0).ops(i).type() == "feed") { - std::string name = prog->blocks(0).ops(i).outputs(0).arguments(0); - inputs_with_no_order.push_back(GetTensorInfo(name, prog->blocks(0))); - int64_t order = -1; - GetOpAttr(prog->blocks(0).ops(i), "col", &order); - input_order.push_back(order); - } - - // This is a trick check, due to the uncorrect shape inference of Paddle - // model - // Remove this after shape inference fixed - if (prog->blocks(0).ops(i).type() == "multiclass_nms3") { - _has_nms = true; - } - } - - // Reorder the inputs and outputs to keep same with the original Paddle model - inputs.resize(input_order.size()); - for (size_t i = 0; i < input_order.size(); ++i) { - inputs[input_order[i]] = inputs_with_no_order[i]; - } - outputs.resize(output_order.size()); - for (size_t i = 0; i < output_order.size(); ++i) { - outputs[output_order[i]] = outputs_with_no_order[i]; - } - - // Trick setting for nms, remove this after shape inference fixed - if (_has_nms) { - for (size_t i = 0; i < outputs.size(); ++i) { - if (outputs[i].shape.size() == 2) { - if (outputs[i].shape[1] == 6) { - outputs[i].shape[0] = -1; - } - } - } - } -} - -int32_t PaddleDataTypeSize(int32_t paddle_dtype) { - Assert(paddle_dtype != FP16, "Float16 is not supported."); - if (paddle_dtype == P2ODataType::BOOL) { - return sizeof(bool); - } else if (paddle_dtype == P2ODataType::INT8) { - return sizeof(int8_t); - } else if (paddle_dtype == P2ODataType::INT16) { - return sizeof(int16_t); - } else if (paddle_dtype == P2ODataType::INT32) { - return sizeof(int32_t); - } else if (paddle_dtype == P2ODataType::INT64) { - return sizeof(int64_t); - } else if (paddle_dtype == P2ODataType::FP32) { - return sizeof(float); - } else if (paddle_dtype == P2ODataType::FP64) { - return sizeof(double); - } else if (paddle_dtype == P2ODataType::UINT8) { - return sizeof(uint8_t); - } else { - Assert(false, "Unexpected data type: " + std::to_string(paddle_dtype)); - } - return -1; -} - -bool PaddleParser::ExistsDumplicateTensorName() const { - std::set names; - for (auto i = 0; i < prog->blocks(0).ops_size(); ++i) { - auto& op = prog->blocks(0).ops(i); - for (auto j = 0; j < op.outputs_size(); ++j) { - for (auto k = 0; k < op.outputs(j).arguments_size(); ++k) { - if (op.type() == "fetch") { - continue; - } - if (names.find(op.outputs(j).arguments(k)) != names.end()) { - P2OLogger() << "There's dumplicate output name: " - << op.outputs(j).arguments(k) - << " in this model, not supported yet." << std::endl; - return true; - } - names.insert(op.outputs(j).arguments(k)); - } - } - } - return false; -} -} // namespace paddle2onnx diff --git a/paddle2onnx/parser/parser.h b/paddle2onnx/parser/parser.h deleted file mode 100644 index 0d15a72bc5d..00000000000 --- a/paddle2onnx/parser/parser.h +++ /dev/null @@ -1,258 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include -#include -#include -#include - -#include "paddle2onnx/proto/p2o_paddle.pb.h" -#include "paddle2onnx/utils/utils.h" - -namespace paddle2onnx { - -enum P2ODataType { - BOOL, - INT16, - INT32, - INT64, - FP16, - FP32, - FP64, - X7, - X8, - X9, - X10, - X11, - X12, - X13, - X14, - X15, - X16, - X17, - X18, - X19, - UINT8, - INT8, -}; -int32_t PaddleDataTypeSize(int32_t paddle_dtype); - -struct TensorInfo { - std::string name; - std::vector shape; - int64_t Rank() const { return static_cast(shape.size()); } - int32_t dtype; - bool is_tensor_array = false; - - TensorInfo() {} - TensorInfo(const std::string& _name, const std::vector& _shape, - const int32_t& _dtype) { - name = _name; - shape.assign(_shape.begin(), _shape.end()); - dtype = _dtype; - } - - TensorInfo(const TensorInfo& info) { - name = info.name; - shape.assign(info.shape.begin(), info.shape.end()); - dtype = info.dtype; - is_tensor_array = info.is_tensor_array; - } -}; - -struct Weight { - std::vector buffer; - std::vector shape; - int32_t dtype; - - template - void set(int32_t data_type, const std::vector& dims, - const std::vector& data) { - buffer.clear(); - shape.clear(); - dtype = data_type; - buffer.resize(data.size() * PaddleDataTypeSize(dtype)); - memcpy(buffer.data(), data.data(), data.size() * PaddleDataTypeSize(dtype)); - for (auto& d : dims) { - shape.push_back(d); - } - } - template - void get(std::vector* data) const { - int64_t nums = std::accumulate(std::begin(shape), std::end(shape), 1, - std::multiplies()); - data->resize(nums); - if (dtype == P2ODataType::INT64) { - std::vector value(nums); - memcpy(value.data(), buffer.data(), nums * sizeof(int64_t)); - data->assign(value.begin(), value.end()); - } else if (dtype == P2ODataType::INT32) { - std::vector value(nums); - memcpy(value.data(), buffer.data(), nums * sizeof(int32_t)); - data->assign(value.begin(), value.end()); - } else if (dtype == P2ODataType::INT8) { - std::vector value(nums); - memcpy(value.data(), buffer.data(), nums * sizeof(int8_t)); - data->assign(value.begin(), value.end()); - } else if (dtype == P2ODataType::FP32) { - std::vector value(nums); - memcpy(value.data(), buffer.data(), nums * sizeof(float)); - data->assign(value.begin(), value.end()); - } else if (dtype == P2ODataType::FP64) { - std::vector value(nums); - memcpy(value.data(), buffer.data(), nums * sizeof(double)); - data->assign(value.begin(), value.end()); - } else { - Assert(false, - "Weight::get() only support int64/int32/int8/float32/float64."); - } - } -}; - -class PaddleParser { - public: - // recording variable name:id for each block of a program - std::vector> _blocks_var_name2id; - // recoring set of operators for each block - std::vector> - _blocks_ops; - std::shared_ptr prog; - std::map params; - std::vector inputs; - std::vector outputs; - bool is_quantized_model = false; // If the Paddle model is a quantized model, - // set is_quantized_model to be true - - bool Init(const std::string& _model, const std::string& _params = ""); - bool Init(const void* model_buffer, int64_t model_size, - const void* params_buffer = nullptr, int64_t params_size = 0); - void InitBlock(); - - int NumOfBlocks() const; - int NumOfOps(int block_idx) const; - bool HasNms() const { return _has_nms; } - const framework::proto::OpDesc& GetOpDesc(int32_t block_idx, - int32_t op_idx) const; - - bool OpHasInput(int64_t block_id, int64_t op_id, - const std::string& name) const; - bool OpHasOutput(int64_t block_id, int64_t op_id, - const std::string& name) const; - - std::vector GetOpInput(int64_t block_id, int64_t op_id, - const std::string& name) const; - std::vector GetOpOutput(int64_t block_id, int64_t op_id, - const std::string& name) const; - - bool OpIsAttrVar(int64_t block_id, int64_t op_id, - const std::string& name) const; - - std::vector GetOpAttrVar(int64_t block_id, int64_t op_id, - const std::string& name) const; - - bool OpHasAttr(const paddle2onnx::framework::proto::OpDesc& op, - const std::string& name) const; - - void GetOpAttr(const paddle2onnx::framework::proto::OpDesc& op, - const std::string& name, int64_t* res) const; - void GetOpAttr(const paddle2onnx::framework::proto::OpDesc& op, - const std::string& name, float* res) const; - void GetOpAttr(const paddle2onnx::framework::proto::OpDesc& op, - const std::string& name, bool* res) const; - void GetOpAttr(const paddle2onnx::framework::proto::OpDesc& op, - const std::string& name, std::string* res) const; - void GetOpAttr(const paddle2onnx::framework::proto::OpDesc& op, - const std::string& name, std::vector* res) const; - void GetOpAttr(const paddle2onnx::framework::proto::OpDesc& op, - const std::string& name, std::vector* res) const; - void GetOpAttr(const paddle2onnx::framework::proto::OpDesc& op, - const std::string& name, std::vector* res) const; - - bool IsConstantTensor(const int64_t& block_idx, - const std::string& tensor_name) const; - template - bool TryGetTensorValue(const int64_t& block_id, - const std::string& tensor_name, - std::vector* data) const; - - private: - // If the model has same output name in difference operators - // will fail to convert - bool IsAttrVar(const paddle2onnx::framework::proto::OpDesc& op, - const int64_t& attr_id) const; - bool ExistsDumplicateTensorName() const; - void GetBlocksVarName2Id(); - void GetBlocksOps(); - TensorInfo GetTensorInfo( - const std::string& name, - const paddle2onnx::framework::proto::BlockDesc& block) const; - void GetGlobalBlockInputOutputInfo(); - bool GetParamNames(std::vector* var_names); - bool LoadProgram(const std::string& model); - bool LoadProgram(const void* model_buffer, int64_t model_size); - bool LoadParams(const std::string& path); - bool LoadParamsFromMemoryBuffer(const std::string& buffer); - bool LoadParamsFromMemoryBuffer(const void* params_buffer, - int64_t params_size); - // This is a trick flag - // While there's a nms operator in paddle model, - // the shape inference of paddle is not correct - bool _has_nms = false; - std::vector> _constant_ops; -}; - -template -bool PaddleParser::TryGetTensorValue(const int64_t& block_id, - const std::string& tensor_name, - std::vector* data) const { - { - auto iter = params.find(tensor_name); - if (iter != params.end()) { - (iter->second).get(data); - return true; - } - } - Assert(block_id < _constant_ops.size(), - "block_id is out of range while calling TryGetTensorValue."); - auto iter = _constant_ops[block_id].find(tensor_name); - if (iter == _constant_ops[block_id].end()) { - return false; - } - Assert(iter->second < _blocks_ops[block_id].size(), - "op_idx is out of range while calling TryGetTensorValue."); - auto op = _blocks_ops[block_id][iter->second]; - int64_t dtype; - GetOpAttr(*op, "dtype", &dtype); - if (dtype == P2ODataType::INT64) { - std::vector value; - GetOpAttr(*op, "int64_values", &value); - data->assign(value.begin(), value.end()); - } else if (dtype == P2ODataType::INT32) { - std::vector value; - GetOpAttr(*op, "int32_values", &value); - data->assign(value.begin(), value.end()); - } else if (dtype == P2ODataType::FP32) { - std::vector value; - GetOpAttr(*op, "fp32_values", &value); - data->assign(value.begin(), value.end()); - } else { - Assert( - false, - "Only support int32/int64/float32 data type in assign_value operator."); - } - return true; -} - -} // namespace paddle2onnx diff --git a/paddle2onnx/proto/CMakeLists.txt b/paddle2onnx/proto/CMakeLists.txt deleted file mode 100644 index 41c2ba81e89..00000000000 --- a/paddle2onnx/proto/CMakeLists.txt +++ /dev/null @@ -1,11 +0,0 @@ -if(NOT TARGET protobuf) - include(FindProtobuf) - find_package(Protobuf REQUIRED) - set(Protobuf_USE_STATIC_LIBS ON) - include_directories(${PROTOBUF_INCLUDE_DIR}) - include_directories(${CMAKE_CURRENT_SOURCE_DIR}) - set(PROTOBUF_LIBRARIES ${PROTOBUF_LIBRARIES} CACHE INTERNAL "" FORCE) -endif() - -PROTOBUF_GENERATE_CPP(PROTO_SRC PROTO_HEADER p2o_paddle.proto) -add_library(p2o_paddle_proto ${PROTO_HEADER} ${PROTO_SRC}) diff --git a/paddle2onnx/proto/p2o_paddle.proto b/paddle2onnx/proto/p2o_paddle.proto deleted file mode 100644 index 187599c7132..00000000000 --- a/paddle2onnx/proto/p2o_paddle.proto +++ /dev/null @@ -1,272 +0,0 @@ -/* Copyright (c) 2016 PaddlePaddle Authors. All Rights Reserved. - -Licensed under the Apache License, Version 2.0 (the "License"); -you may not use this file except in compliance with the License. -You may obtain a copy of the License at - - http://www.apache.org/licenses/LICENSE-2.0 - -Unless required by applicable law or agreed to in writing, software -distributed under the License is distributed on an "AS IS" BASIS, -WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -See the License for the specific language governing permissions and -limitations under the License. */ - -syntax = "proto2"; -package paddle2onnx.framework.proto; - -// Any incompatible changes to ProgramDesc and its dependencies should -// raise the version defined version.h. -// -// Serailization and Deserialization codes should be modified in a way -// that supports old versions following the version and compatibility policy. -message Version { optional int64 version = 1 [ default = 0 ]; } - -enum AttrType { - INT = 0; - FLOAT = 1; - STRING = 2; - INTS = 3; - FLOATS = 4; - STRINGS = 5; - BOOLEAN = 6; - BOOLEANS = 7; - BLOCK = 8; - LONG = 9; - BLOCKS = 10; - LONGS = 11; - FLOAT64S = 12; - VAR = 13; - VARS = 14; - FLOAT64 = 15; - SCALAR = 16; - SCALARS = 17; -} - - -message Complex { - required double r = 1; - required double i = 2; -}; - -message Scalar { - enum Type { - BOOLEAN = 1; - LONG = 2; - FLOAT64 = 3; - COMPLEX128 = 4; - } - required Type type = 1; - - optional bool b = 2; - optional int64 i = 3; - optional double r = 4; - optional Complex c = 5; -}; - -// OpDesc describes an instance of a C++ framework::OperatorBase -// derived class type. -message OpDesc { - - message Attr { - required string name = 1; - required AttrType type = 2; - optional int32 i = 3; - optional float f = 4; - optional string s = 5; - repeated int32 ints = 6; - repeated float floats = 7; - repeated string strings = 8; - optional bool b = 10; - repeated bool bools = 11; - optional int32 block_idx = 12; - optional int64 l = 13; - repeated int32 blocks_idx = 14; - repeated int64 longs = 15; - repeated double float64s = 16; - optional string var_name = 17; - repeated string vars_name = 18; - optional double float64 = 19; - optional Scalar scalar = 20; - repeated Scalar scalars = 21; - }; - - message Var { - required string parameter = 1; - repeated string arguments = 2; - }; - - required string type = 3; - repeated Var inputs = 1; - repeated Var outputs = 2; - repeated Attr attrs = 4; - optional bool is_target = 5 [ default = false ]; -}; - -// OpProto describes a C++ framework::OperatorBase derived class. -message OpProto { - - // VarProto describes the C++ type framework::Variable. - message Var { - required string name = 1; - required string comment = 2; - - optional bool duplicable = 3 [ default = false ]; - optional bool intermediate = 4 [ default = false ]; - optional bool dispensable = 5 [ default = false ]; - optional bool extra = 6 [ default = false ]; - optional bool quant = 7 [ default = false ]; - } - - // AttrProto describes the C++ type Attribute. - message Attr { - required string name = 1; - required AttrType type = 2; - required string comment = 3; - // If that attribute is generated, it means the Paddle third - // language binding has responsibility to fill that - // attribute. End-User should not set that attribute. - optional bool generated = 4 [ default = false ]; - optional bool extra = 5 [ default = false ]; - optional bool quant = 6 [ default = false ]; - optional bool support_tensor = 7 [ default = false]; - } - - required string type = 1; - repeated Var inputs = 2; - repeated Var outputs = 3; - repeated Attr attrs = 4; - required string comment = 5; -} - -message VarType { - enum Type { - // Pod Types - BOOL = 0; - INT16 = 1; - INT32 = 2; - INT64 = 3; - FP16 = 4; - FP32 = 5; - FP64 = 6; - // phi::DenseTensor is used in C++. - SIZE_T = 19; - UINT8 = 20; - INT8 = 21; - BF16 = 22; - COMPLEX64 = 23; - COMPLEX128 = 24; - - // Other types that may need additional descriptions - LOD_TENSOR = 7; - SELECTED_ROWS = 8; - FEED_MINIBATCH = 9; - FETCH_LIST = 10; - STEP_SCOPES = 11; - LOD_RANK_TABLE = 12; - LOD_TENSOR_ARRAY = 13; - PLACE_LIST = 14; - READER = 15; - // Any runtime decided variable type is raw - // raw variables should manage their own allocations - // in operators like nccl_op - RAW = 17; - TUPLE = 18; - - STRING = 25; - STRINGS = 26; - VOCAB = 27; - FEED_LIST = 28; - // The data type of phi::StringTensor - PSTRING = 29; - // the data type of phi::SparseCooTensor - SPARSE_COO = 30; - // the data type of phi::SparseCsrTensor - SPARSE_CSR = 31; - } - - required Type type = 1; - - message TensorDesc { - // Should only be PODType. Is enforced in C++ - required Type data_type = 1; - repeated int64 dims = 2; // [UNK, 640, 480] is saved as [-1, 640, 480] - } - optional TensorDesc selected_rows = 2; - - message LoDTensorDesc { - required TensorDesc tensor = 1; - optional int32 lod_level = 2 [ default = 0 ]; - } - optional LoDTensorDesc lod_tensor = 3; - - message LoDTensorArrayDesc { - required TensorDesc tensor = 1; - optional int32 lod_level = 2 [ default = 0 ]; - } - optional LoDTensorArrayDesc tensor_array = 4; - - message ReaderDesc { repeated LoDTensorDesc lod_tensor = 1; } - optional ReaderDesc reader = 5; - - message Tuple { repeated Type element_type = 1; } - optional Tuple tuple = 7; - - optional TensorDesc string = 8; - optional TensorDesc strings = 9; - optional TensorDesc vocab = 10; - optional TensorDesc sparse_coo = 11; - optional TensorDesc sparse_csr = 12; -} - -message VarDesc { - - message Attr { - required string name = 1; - required AttrType type = 2; - optional int32 i = 3; - optional string s = 4; - repeated int32 ints = 5; - }; - - required string name = 1; - required VarType type = 2; - optional bool persistable = 3 [ default = false ]; - // True if the variable is an input data and - // have to check the feed data shape and dtype - optional bool need_check_feed = 4 [ default = false ]; - optional bool is_parameter = 5 [ default = false ]; - optional bool stop_gradient = 6 [ default = false ]; - repeated Attr attrs = 7; -} - -message BlockDesc { - required int32 idx = 1; - required int32 parent_idx = 2; - repeated VarDesc vars = 3; - repeated OpDesc ops = 4; - optional int32 forward_block_idx = 5 [ default = -1 ]; -} - -// In some cases, Paddle may perform operator definition iterations, -// and the operator uses OpVersionMap for compatibility testing. -message OpVersion { required int32 version = 1; } -message OpVersionMap { - message OpVersionPair { - required string op_name = 1; - required OpVersion op_version = 2; - } - repeated OpVersionPair pair = 1; -} - -// Please refer to -// https://github.com/PaddlePaddle/Paddle/blob/develop/doc/design/program.md -// for more details. -// TODO(panyx0718): A model can have multiple programs. Need a -// way to distinguish them. Maybe ID or name? -message ProgramDesc { - reserved 2, 3; // For backward compatibility. - repeated BlockDesc blocks = 1; - optional Version version = 4; - optional OpVersionMap op_version_map = 5; -} diff --git a/paddle2onnx/utils.py b/paddle2onnx/utils.py deleted file mode 100644 index a65ca483b75..00000000000 --- a/paddle2onnx/utils.py +++ /dev/null @@ -1,131 +0,0 @@ -# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from __future__ import absolute_import - -import importlib -import collections -import time -import os -import sys - - -def try_import(module_name): - """Try importing a module, with an informative error message on failure.""" - install_name = module_name - try: - mod = importlib.import_module(module_name) - return mod - except ImportError: - err_msg = ( - "Failed importing {}. This likely means that some modules " - "requires additional dependencies that have to be " - "manually installed (usually with `pip install {}`). ").format( - module_name, install_name) - raise ImportError(err_msg) - - -def check_model(onnx_model): - onnx = try_import('onnx') - try: - onnx.checker.check_model(onnx_model) - except Exception: - raise Exception('ONNX model is not valid.') - finally: - logging.info('ONNX model generated is valid.') - - -levels = {0: 'ERROR', 1: 'WARNING', 2: 'INFO', 3: 'DEBUG'} - - -class logging(): - log_level = 2 - - @staticmethod - def log(level=2, message="", use_color=False): - current_time = time.time() - time_array = time.localtime(current_time) - current_time = time.strftime("%Y-%m-%d %H:%M:%S", time_array) - if logging.log_level >= level: - if use_color: - print("\033[1;31;40m{} [{}]\t{}\033[0m".format( - current_time, levels[level], message).encode("utf-8") - .decode("latin1")) - else: - print("{} [{}]\t{}".format(current_time, levels[level], message) - .encode("utf-8").decode("latin1")) - sys.stdout.flush() - - @staticmethod - def debug(message="", use_color=False): - logging.log(level=3, message=message, use_color=use_color) - - @staticmethod - def info(message="", use_color=False): - logging.log(level=2, message=message, use_color=use_color) - - @staticmethod - def warning(message="", use_color=True): - logging.log(level=1, message=message, use_color=use_color) - - @staticmethod - def error(message="", use_color=True, exit=True): - logging.log(level=0, message=message, use_color=use_color) - if exit: - sys.exit(-1) - - -def compare_value(a, b, cond): - if cond == 'equal': - if a != b: - return False - return True - if cond == 'greater_than': - if a <= b: - return False - return True - if cond == 'greater_equal': - if a < b: - return False - return True - if cond == 'less_equal': - if a > b: - return False - return True - if cond == 'less_than': - if a >= b: - return False - return True - - -def compare_attr(actual_value, target_value, attr_name, cond='equal'): - if not compare_value(actual_value, target_value, cond): - raise ValueError('Support {} {} {}, actually got {}=={}.'.format( - attr_name, cond, target_value, attr_name, actual_value)) - - -def compare_attr_between_dims(attr, dims, attr_name, cond='equal'): - if not compare_value(attr[dims[0]], attr[dims[1]], cond): - expect_info = 'Support {}[{}] {} {}[{}], '.format( - attr_name, dims[0], cond, attr_name, dims[1]) - actual_info = 'actually got {}[{}]=={}, not {} {}[{}]=={}.'.format( - attr_name, dims[0], attr[dims[0]], cond, attr_name, dims[1], - attr[dims[1]]) - raise ValueError(expect_info + actual_info) - - -def require_fixed_shape(op_name=None): - logging.error( - "[{}]Fixed shape is required, refer this doc for more information: https://github.com/PaddlePaddle/Paddle2ONNX/blob/develop/docs/zh/FAQ.md". - format(op_name)) diff --git a/paddle2onnx/utils/utils.h b/paddle2onnx/utils/utils.h deleted file mode 100644 index 463760703bf..00000000000 --- a/paddle2onnx/utils/utils.h +++ /dev/null @@ -1,80 +0,0 @@ -// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#pragma once -#include - -#include -#include -#include - -namespace paddle2onnx { - -inline void Assert(bool condition, const std::string& message) { - if (!condition) { - fprintf(stderr, "[ERROR] %s\n", message.c_str()); - std::abort(); - } -} - -inline const std::string RequireOpset(const int32_t& opset_version) { - return "Requires the minimal opset version of " + - std::to_string(opset_version) + "."; -} - -class P2OLogger { - public: - P2OLogger() { - line_ = ""; - prefix_ = "[Paddle2ONNX]"; - verbose_ = true; - } - explicit P2OLogger(bool verbose, - const std::string& prefix = "[Paddle2ONNX]") { - verbose_ = verbose; - line_ = ""; - prefix_ = prefix; - } - - template - P2OLogger& operator<<(const T& val) { - if (!verbose_) { - return *this; - } - std::stringstream ss; - ss << val; - line_ += ss.str(); - return *this; - } - P2OLogger& operator<<(std::ostream& (*os)(std::ostream&)) { - if (!verbose_) { - return *this; - } - std::cout << prefix_ << " " << line_ << std::endl; - line_ = ""; - return *this; - } - ~P2OLogger() { - if (!verbose_ && line_ != "") { - std::cout << line_ << std::endl; - } - } - - private: - std::string line_; - std::string prefix_; - bool verbose_ = true; -}; - -} // namespace paddle2onnx diff --git a/poros/CMakeLists.txt b/poros/CMakeLists.txt deleted file mode 100755 index b332f85c775..00000000000 --- a/poros/CMakeLists.txt +++ /dev/null @@ -1,217 +0,0 @@ -cmake_minimum_required(VERSION 3.21) -project(poros) -set(CMAKE_CXX_STANDARD 14) - -option(BUILD_STATIC "build lib${PROJECT_NAME}.a static lib" OFF) -option(BUILD_KERNEL "build lib${PROJECT_NAME}-kernel.so shared lib" OFF) -option(BUILD_STATIC_KERNEL "build lib${PROJECT_NAME}-kernel.a static lib" OFF) -option(BUILD_TOOL "build ${PROJECT_NAME}-tool, an executable binary output" OFF) -option(TEST "build for test. copy '.so' to site-packages automatically after compile" OFF) -option(DEBUG "build for debug. add '-g' flag to gcc for detailed debug information" ON) -option(UT "build for unit test" OFF) -option(ABI "build ${PROJECT_NAME} with application binary interface (ABI) is on" OFF) - -# abi configuration -if (NOT ABI) - add_definitions(-D_GLIBCXX_USE_CXX11_ABI=0) -else () - if (BUILD_TOOL OR UT) - message(FATAL_ERROR "${PROJECT_NAME}-tool or unit test are not supported when abi is on.") - endif() - add_definitions(-D_GLIBCXX_USE_CXX11_ABI=1) -endif () - -# minimum requirements -set(PYTHON_MINIMUM_VERSION 3.6) -set(CUDA_MINIMUM_VERSION 10.2) -set(PYTORCH_MINIMUM_VERSION 1.9) -set(TENSORRT_MINIMUM_VERSION 8.2) -set(CUDNN_MINIMUM_VERSION 8.0) - -list(APPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/cmake") - -# find cuda -find_package(CUDAToolkit ${CUDA_MINIMUM_VERSION} REQUIRED) - -# find python3 -find_package(Python3 ${PYTHON_MINIMUM_VERSION} REQUIRED COMPONENTS Interpreter Development) -message(STATUS "Found Python: ${Python3_VERSION_MAJOR}.${Python3_VERSION_MINOR}.${Python3_VERSION_PATCH}") - -if (NOT Python3_SITELIB) - message(FATAL_ERROR "site-packages not found. ") -else () - message(STATUS "site-packages: ${Python3_SITELIB}") -endif () - -# find pytorch -find_package(Torch ${PYTORCH_MINIMUM_VERSION} REQUIRED HINTS ${Python3_SITELIB}) - -# find TensorRT -find_package(TensorRT ${TENSORRT_MINIMUM_VERSION} REQUIRED) -get_filename_component(TENSORRT_LIB_DIR ${TensorRT_LIBRARIES} DIRECTORY) - - -# find CUDNN -find_package(CUDNN ${CUDNN_MINIMUM_VERSION} REQUIRED) - - -## release headers -# engine -file(GLOB headers "${PROJECT_SOURCE_DIR}/poros/engine/iengine.h" "${PROJECT_SOURCE_DIR}/poros/engine/engine_context.h") -file(COPY ${headers} DESTINATION "${PROJECT_SOURCE_DIR}/build/include/poros/engine") -# compile -file(GLOB headers "${PROJECT_SOURCE_DIR}/poros/compile/poros_module.h") -file(COPY ${headers} DESTINATION "${PROJECT_SOURCE_DIR}/build/include/poros/compile") -file(GLOB headers "${PROJECT_SOURCE_DIR}/poros/compile/compile.h") -file(COPY ${headers} DESTINATION "${PROJECT_SOURCE_DIR}/build/include/poros/compile") -# converter -file(GLOB headers "${PROJECT_SOURCE_DIR}/poros/converter/iconverter.h") -file(COPY ${headers} DESTINATION "${PROJECT_SOURCE_DIR}/build/include/poros/converter") -# iplugin -file(GLOB headers "${PROJECT_SOURCE_DIR}/poros/iplugin/*.h") -file(COPY ${headers} DESTINATION "${PROJECT_SOURCE_DIR}/build/include/poros/iplugin") -## context -file(GLOB headers "${PROJECT_SOURCE_DIR}/poros/context/*.h") -file(COPY ${headers} DESTINATION "${PROJECT_SOURCE_DIR}/build/include/poros/context") -## context -file(GLOB headers "${PROJECT_SOURCE_DIR}/poros/context/*.h") -file(COPY ${headers} DESTINATION "${PROJECT_SOURCE_DIR}/build/include/poros/context") -## lowering -file(GLOB headers "${PROJECT_SOURCE_DIR}/poros/lowering/*.h") -file(COPY ${headers} DESTINATION "${PROJECT_SOURCE_DIR}/build/include/poros/lowering") -## util -file(GLOB headers "${PROJECT_SOURCE_DIR}/poros/util/*.h") -file(COPY ${headers} DESTINATION "${PROJECT_SOURCE_DIR}/build/include/poros/util") -## log -file(GLOB headers "${PROJECT_SOURCE_DIR}/poros/log/*.h") -file(COPY ${headers} DESTINATION "${PROJECT_SOURCE_DIR}/build/include/poros/log") - - -include_directories(${TORCH_INCLUDE_DIRS}) -include_directories(${TensorRT_INCLUDE_DIRS}) -include_directories(${CUDA_INCLUDE_DIRS}) -include_directories(${CUDNN_INCLUDE_PATH}) -include_directories(${PROJECT_SOURCE_DIR}) -include_directories(poros/compile) - - -add_compile_options(-D__const__= -D_GNU_SOURCE) -add_compile_options(-lpthread -lcrypto -lrt -ldl -lz -fPIC -rdynamic) -add_compile_options(-std=c++17 -O2 -g -pipe -W -Wall -fPIC -Wno-deprecated-declarations -Wno-unused-parameter) -if (DEBUG) - add_compile_options(-g) # for debug -endif () - -add_compile_options( - -Wall - -Wno-comment - -Wno-error=implicit-fallthrough - -Wno-error=unused-but-set-variable - -Wno-error=misleading-indentation - -Wno-error=unused-function - -Wno-error=terminate - -Wno-unused-parameter - -Wno-deprecated-declarations -) - - -file( - GLOB POROS_CPP_FILES - "./poros/*/*.cpp" - "./poros/converter/*/*.cpp" - "./poros/converter/gpu/plugins/*.cpp" -) - - -# libporos.so -add_library(${PROJECT_NAME} SHARED ${POROS_CPP_FILES}) - -#set_target_properties(${PROJECT_NAME} PROPERTIES LIBRARY_OUTPUT_DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR}/python/poros/lib) -set_target_properties(${PROJECT_NAME} PROPERTIES LIBRARY_OUTPUT_DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR}/build/lib) - -#add_custom_command( -# TARGET ${PROJECT_NAME} -# COMMAND ${CMAKE_COMMAND} -E copy ${CMAKE_CURRENT_SOURCE_DIR}/output/lib/lib${PROJECT_NAME}.so ${CMAKE_CURRENT_SOURCE_DIR}/python/poros/lib/ -#) - -# copy libporos.so to python site-packages/poros/lib for testing -if (TEST) - add_custom_command( - TARGET ${PROJECT_NAME} - POST_BUILD - COMMENT "copy ${LIBRARY_OUTPUT_PATH}/lib${PROJECT_NAME}.so to ${Python3_SITELIB}/poros/lib/lib${PROJECT_NAME}.so for rapid testing" - COMMAND ${CMAKE_COMMAND} -E copy ${LIBRARY_OUTPUT_DIRECTORY}/lib${PROJECT_NAME}.so ${Python3_SITELIB}/poros/lib/lib${PROJECT_NAME}.so - ) -endif () - - -if (BUILD_STATIC) - add_library(${PROJECT_NAME}-static STATIC ${POROS_CPP_FILES}) - set_target_properties(${PROJECT_NAME}-static PROPERTIES OUTPUT_NAME ${PROJECT_NAME}) -endif () - -# build gflags -set(GFLAGS_NAMESPACE google) -add_subdirectory(third_party/gflags) - - -find_package(BLAS) - -add_custom_target( - Creating_Symlink ALL - COMMAND_EXPAND_LISTS - COMMAND ${CMAKE_COMMAND} -E make_directory ${CMAKE_CURRENT_SOURCE_DIR}/third_party - COMMAND ${CMAKE_COMMAND} -E create_symlink ${TENSORRT_LIB_DIR} ${CMAKE_CURRENT_SOURCE_DIR}/third_party/tensorrtlib - WORKING_DIRECTORY ${CMAKE_CURRENT_BINARY_DIR} - COMMENT "Creating Symlink ${CMAKE_CURRENT_SOURCE_DIR}/third_party/tensorrtlib -> ${TENSORRT_LIB_DIR} " - VERBATIM -) - -# executable -if (BUILD_TOOL) - set(POROS_TOOL ${PROJECT_NAME}-tool) - - - add_executable(${POROS_TOOL}) - - target_sources(${POROS_TOOL} PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/tools/main.cpp) - target_sources(${POROS_TOOL} PUBLIC ${POROS_CPP_FILES}) - - target_link_libraries(${POROS_TOOL} gflags::gflags) - target_link_libraries(${POROS_TOOL} TensorRT::TensorRT TensorRT::Plugin) - target_link_libraries(${POROS_TOOL} torch) -# target_link_libraries(${POROS_TOOL} CUDA::toolkit) - target_link_libraries(${POROS_TOOL} CUDA::cudart CUDA::cusolver CUDA::cublas CUDA::cusolver CUDA::cusparse) - target_link_libraries(${POROS_TOOL} BLAS::BLAS) - -endif () - - - -# kernel -file( - GLOB POROS_KERNEL_CPP_FILES - ./poros/compile/*.cpp - ./poros/context/*.cpp - ./poros/iplugin/*.cpp - ./poros/log/*.cpp - ./poros/lowering/*.cpp - ./poros/util/*.cpp - ./poros/engine/engine.cpp -) - -# kernel SHARED -if (BUILD_KERNEL) - add_library(${PROJECT_NAME}-kernel SHARED ${POROS_KERNEL_CPP_FILES}) -endif () - -# kernel STATIC -if (BUILD_STATIC_KERNEL) - add_library(${PROJECT_NAME}-kernel-static STATIC ${POROS_KERNEL_CPP_FILES}) - set_target_properties(${PROJECT_NAME}-kernel-static PROPERTIES OUTPUT_NAME ${PROJECT_NAME}-kernel) -endif () - -if (UT) - add_subdirectory(third_party/googletest) - add_subdirectory(unittest) -endif () diff --git a/poros/LICENSE b/poros/LICENSE deleted file mode 100644 index f58676f3e20..00000000000 --- a/poros/LICENSE +++ /dev/null @@ -1,203 +0,0 @@ -Copyright (c) 2022 Baidu, Inc. All Rights Reserved. - - Apache License - Version 2.0, January 2004 - http://www.apache.org/licenses/ - - TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION - - 1. Definitions. - - "License" shall mean the terms and conditions for use, reproduction, - and distribution as defined by Sections 1 through 9 of this document. - - "Licensor" shall mean the copyright owner or entity authorized by - the copyright owner that is granting the License. - - "Legal Entity" shall mean the union of the acting entity and all - other entities that control, are controlled by, or are under common - control with that entity. For the purposes of this definition, - "control" means (i) the power, direct or indirect, to cause the - direction or management of such entity, whether by contract or - otherwise, or (ii) ownership of fifty percent (50%) or more of the - outstanding shares, or (iii) beneficial ownership of such entity. - - "You" (or "Your") shall mean an individual or Legal Entity - exercising permissions granted by this License. - - "Source" form shall mean the preferred form for making modifications, - including but not limited to software source code, documentation - source, and configuration files. - - "Object" form shall mean any form resulting from mechanical - transformation or translation of a Source form, including but - not limited to compiled object code, generated documentation, - and conversions to other media types. - - "Work" shall mean the work of authorship, whether in Source or - Object form, made available under the License, as indicated by a - copyright notice that is included in or attached to the work - (an example is provided in the Appendix below). - - "Derivative Works" shall mean any work, whether in Source or Object - form, that is based on (or derived from) the Work and for which the - editorial revisions, annotations, elaborations, or other modifications - represent, as a whole, an original work of authorship. For the purposes - of this License, Derivative Works shall not include works that remain - separable from, or merely link (or bind by name) to the interfaces of, - the Work and Derivative Works thereof. - - "Contribution" shall mean any work of authorship, including - the original version of the Work and any modifications or additions - to that Work or Derivative Works thereof, that is intentionally - submitted to Licensor for inclusion in the Work by the copyright owner - or by an individual or Legal Entity authorized to submit on behalf of - the copyright owner. For the purposes of this definition, "submitted" - means any form of electronic, verbal, or written communication sent - to the Licensor or its representatives, including but not limited to - communication on electronic mailing lists, source code control systems, - and issue tracking systems that are managed by, or on behalf of, the - Licensor for the purpose of discussing and improving the Work, but - excluding communication that is conspicuously marked or otherwise - designated in writing by the copyright owner as "Not a Contribution." - - "Contributor" shall mean Licensor and any individual or Legal Entity - on behalf of whom a Contribution has been received by Licensor and - subsequently incorporated within the Work. - - 2. Grant of Copyright License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - copyright license to reproduce, prepare Derivative Works of, - publicly display, publicly perform, sublicense, and distribute the - Work and such Derivative Works in Source or Object form. - - 3. Grant of Patent License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - (except as stated in this section) patent license to make, have made, - use, offer to sell, sell, import, and otherwise transfer the Work, - where such license applies only to those patent claims licensable - by such Contributor that are necessarily infringed by their - Contribution(s) alone or by combination of their Contribution(s) - with the Work to which such Contribution(s) was submitted. If You - institute patent litigation against any entity (including a - cross-claim or counterclaim in a lawsuit) alleging that the Work - or a Contribution incorporated within the Work constitutes direct - or contributory patent infringement, then any patent licenses - granted to You under this License for that Work shall terminate - as of the date such litigation is filed. - - 4. Redistribution. You may reproduce and distribute copies of the - Work or Derivative Works thereof in any medium, with or without - modifications, and in Source or Object form, provided that You - meet the following conditions: - - (a) You must give any other recipients of the Work or - Derivative Works a copy of this License; and - - (b) You must cause any modified files to carry prominent notices - stating that You changed the files; and - - (c) You must retain, in the Source form of any Derivative Works - that You distribute, all copyright, patent, trademark, and - attribution notices from the Source form of the Work, - excluding those notices that do not pertain to any part of - the Derivative Works; and - - (d) If the Work includes a "NOTICE" text file as part of its - distribution, then any Derivative Works that You distribute must - include a readable copy of the attribution notices contained - within such NOTICE file, excluding those notices that do not - pertain to any part of the Derivative Works, in at least one - of the following places: within a NOTICE text file distributed - as part of the Derivative Works; within the Source form or - documentation, if provided along with the Derivative Works; or, - within a display generated by the Derivative Works, if and - wherever such third-party notices normally appear. The contents - of the NOTICE file are for informational purposes only and - do not modify the License. You may add Your own attribution - notices within Derivative Works that You distribute, alongside - or as an addendum to the NOTICE text from the Work, provided - that such additional attribution notices cannot be construed - as modifying the License. - - You may add Your own copyright statement to Your modifications and - may provide additional or different license terms and conditions - for use, reproduction, or distribution of Your modifications, or - for any such Derivative Works as a whole, provided Your use, - reproduction, and distribution of the Work otherwise complies with - the conditions stated in this License. - - 5. Submission of Contributions. Unless You explicitly state otherwise, - any Contribution intentionally submitted for inclusion in the Work - by You to the Licensor shall be under the terms and conditions of - this License, without any additional terms or conditions. - Notwithstanding the above, nothing herein shall supersede or modify - the terms of any separate license agreement you may have executed - with Licensor regarding such Contributions. - - 6. Trademarks. This License does not grant permission to use the trade - names, trademarks, service marks, or product names of the Licensor, - except as required for reasonable and customary use in describing the - origin of the Work and reproducing the content of the NOTICE file. - - 7. Disclaimer of Warranty. Unless required by applicable law or - agreed to in writing, Licensor provides the Work (and each - Contributor provides its Contributions) on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or - implied, including, without limitation, any warranties or conditions - of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A - PARTICULAR PURPOSE. You are solely responsible for determining the - appropriateness of using or redistributing the Work and assume any - risks associated with Your exercise of permissions under this License. - - 8. Limitation of Liability. In no event and under no legal theory, - whether in tort (including negligence), contract, or otherwise, - unless required by applicable law (such as deliberate and grossly - negligent acts) or agreed to in writing, shall any Contributor be - liable to You for damages, including any direct, indirect, special, - incidental, or consequential damages of any character arising as a - result of this License or out of the use or inability to use the - Work (including but not limited to damages for loss of goodwill, - work stoppage, computer failure or malfunction, or any and all - other commercial damages or losses), even if such Contributor - has been advised of the possibility of such damages. - - 9. Accepting Warranty or Additional Liability. While redistributing - the Work or Derivative Works thereof, You may choose to offer, - and charge a fee for, acceptance of support, warranty, indemnity, - or other liability obligations and/or rights consistent with this - License. However, in accepting such obligations, You may act only - on Your own behalf and on Your sole responsibility, not on behalf - of any other Contributor, and only if You agree to indemnify, - defend, and hold each Contributor harmless for any liability - incurred by, or claims asserted against, such Contributor by reason - of your accepting any such warranty or additional liability. - - END OF TERMS AND CONDITIONS - - APPENDIX: How to apply the Apache License to your work. - - To apply the Apache License to your work, attach the following - boilerplate notice, with the fields enclosed by brackets "[]" - replaced with your own identifying information. (Don't include - the brackets!) The text should be enclosed in the appropriate - comment syntax for the file format. We also recommend that a - file or class name and description of purpose be included on the - same "printed page" as the copyright notice for easier - identification within third-party archives. - -Copyright (c) 2022 Baidu, Inc. All Rights Reserved. - - Licensed under the Apache License, Version 2.0 (the "License"); - you may not use this file except in compliance with the License. - You may obtain a copy of the License at - - http://www.apache.org/licenses/LICENSE-2.0 - - Unless required by applicable law or agreed to in writing, software - distributed under the License is distributed on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - See the License for the specific language governing permissions and - limitations under the License. \ No newline at end of file diff --git a/poros/README.md b/poros/README.md deleted file mode 100644 index 7eaacafba55..00000000000 --- a/poros/README.md +++ /dev/null @@ -1,203 +0,0 @@ -# Poros AI Compiler - -## Description - -Poros is an AI Compiler for deep learning framework. It can provide significantly lower inference latency comparing with original model, and provide much flexibility for dynamic graphs. -Poros mainly works on the TorchScript IR currently, that means it supports the models from PyTorch, ONNX, TensorFlow and any other framework that can be converted to TorchScript. Also, we are planning to support more IRs in the future. -Poros is designed to supports multiple hardware backends conveniently. For now, Poros has supported GPU and XPU (BAIDU-Kunlun) Device. It's welcomed to add additional devices. - -## How It Works - -Figure 1 is the architecture of Poros. The central part marked by the red dotted line is Model Optimizer, the main module of Poros. IR graphs are optimized by IR lowering, op fusing, op converting and auto-tuning, and then segmented into engine related subgraph by maximize the op nums of each engine kernel and minimize the total count of engine kernels. - -![image](https://user-images.githubusercontent.com/54064850/203691621-e75d7c17-320c-4dff-8abe-58c3c9db99a2.png) - -In order to achieve the above goals on GPU, we've rewritten hundreds of TorchScript OPs, which reduced extra subgraphs caused by unsupported op during subgraph partitioning. Dozens of lowering strategy including op fusions were employed to reduce the actual calculating load of CUDA Kernels. - -## Dependencies - -Poros is developed based on PyTorch, CUDA, TensorRT (TRT Engine), CuDNN. The minimum_required (recommended) versions of -these packages are listed as below: - -| Package | Minimum Version | Recommended Version | -|----------|-----------------|---------------------| -| PyTorch | 1.9.0 | 1.12.1 | -| CUDA | 10.2 | 11.3 | -| TensorRT | 8.2 | 8.4 | -| CuDNN | 7.6.5 | 8.4 | -| Python | 3.6.5 | 3.8 | - -If you want to build for GPU Inference, it's better to align the CUDA version with the version that PyTorch built on. -For example, we recommend you to use CUDA 11.1+ if the installed PyTorch version is 1.11.0+cu111, or some "undefined -reference CUDA...." errors may appear during building. - -> There is a known cuBlas related issue of CUDA 10.2. If you are using CUDA 10.2, make sure these two patches have be installed. -> https://developer.nvidia.com/cuda-10.2-download-archive?target_os=Linux&target_arch=x86_64&target_distro=Ubuntu&target_version=1804&target_type=runfilelocal - -## How To Build - -### 0. Install Dependencies - -get Poros source code: - -```shell -git clone https://github.com/PaddlePaddle/FastDeploy.git -cd poros -git submodule update --init --recursive -``` - -We strongly recommend you to prepare the building environment with anaconda3: - -```shell -conda create --name poros python=3.8 -conda activate poros -export CMAKE_PREFIX_PATH=$CONDA_PREFIX -conda install cmake==3.22.1 pytorch==1.12.1 cudatoolkit=11.3 numpy -c pytorch -``` -**If CUDA has been installed as system driver, cudatoolkit is not necessary. And CMake version requires >= 3.21, GCC version requires >= 8.2.** - - -Poros uses cmake to manage dependencies. It will find all dependency packages automatically as long as the packages were -installed to the usual location. Otherwise, you should assign the install location of these packages manually. - -```shell -export CUDAToolkit_ROOT=/cuda/install/dir/ #point CUDAToolkit_ROOT to the CUDA installation dir -export TENSORRT_ROOT=/tensorrt/install/dir/ #download from Nvidia and upack, no need to install into system -export CUDNN_ROOT=/cudnn/install/dir/ #download from Nvidia and upack, no need to install into system -``` -Add cuda, tensorrt and cudnn into your environment variables. - -```shell -export PATH=$CUDAToolkit_ROOT/bin:$PATH -export LD_LIBRARY_PATH=$CUDAToolkit_ROOT/lib64:$TENSORRT_ROOT/lib:$CUDNN_ROOT/lib:$LD_LIBRARY_PATH -``` - -Additional dependency `mkl` is needed while building with PyTorch1.11 + CUDA11.1 -It can be added into cmake by installing, if not, you can try to add it by: -```shell -conda install mkl -``` - -Other packages that Poros depend on are: gflags, googletest etc. , they can be downloaded -by ` git submodule update --init --recursive --jobs 0 -f` - -### 1. Build Project with CMake - -```shell -cd poros -mkdir build -cd build -cmake .. -make -``` - -By default, only the shared library (libporos.so) will be built. - -**To build a static lib (libporos.a):** - -```shell -cmake -DBUILD_STATIC=on .. -make -``` - -Poros `kernel` contains the framework of Poros, as well as the IR lowering strategy, the sub-graph segmentation strategy -and the engine manager without any specific engine (e.g. TensorRT). For Developers who want to use their own -engines, `kernel` can be built separately with options as below: - -**To build a shared kernel lib (libporos-kernel.so):** - -```shell -cmake -DBUILD_KERNEL=on .. -make -``` - -**To build a static kernel lib (libporos-kernel.a):** - -```shell -cmake -DBUILD_STATIC_KERNEL=on .. -make -``` - -### 2. Build Distributing Package with setuptools (Python3) - -After the libporos.so has been built, you can build the `.whl` package for Python3: - -```shell -cd ../python -python3 setup.py bdist_wheel -``` - -The output looks like: `poros-0.1.0-cp38-cp38m-linux_x86_64.whl`. It can be installed easily with pip: - -```shell -cd dist -pip3 install poros-0.1.0-cp38-cp38m-linux_x86_64.whl -``` -or, you can use `python3 setup.py develop` to create symbolic link to `python` dir. - -### 3. Build Executable Binary - -We provide an example C++ shell for users who want to build an executable binary. The `main.cpp` file locates -at `tools/main.cpp`, you modify the code according to your needs. The executable binary `poros-tool` can be built with -this command: - -```shell -mkdir build -cd build -cmake -DBUILD_TOOL=on .. -make -``` - -### 4. Build Test -```shell -cmake -DUT=on .. -make -./unit_test # run unit test -``` - - -## How To Use - -### 1. Python Usage: - -```python -import poros -import torch -from torchvision import models - -original_model = models.resnet50(pretrained=True).cuda().eval() #load/download pre-trained model -option = poros.PorosOptions() #set poros option -poros_model = poros.compile(torch.jit.script(original_model), input_datas, option) #build the model - -input = torch.randn(1,3,224,224, dtype=torch.float32).cuda() -poros_res = poros_model(input) # use compiled model in the same way as the original model - -``` - -The complete benchmark example (resnet50) .py script is `python/example/test_resnet.py` - -```shell -python3 python/example/test_resnet.py -``` - -### 2. CPP Usage: - -If the executable binary `poros-tool` is built, you can run the benchmark like this: - -```shell -./poros-tool --module_file_path ../../poros/tools/std_pretrained_resnet50_gpu.pt --test_mode=original #original PyTorch model -./poros-tool --module_file_path ../../poros/tools/std_pretrained_resnet50_gpu.pt --test_mode=poros #poros compiled model -``` -> PyTorch has changed the packaging format of model since 1.4+, while the pretrained model of resnet50 is still using the old format (.tar). -> You may need to convert the format to the newer one (.zip) by your self. Convert command like this: -> ```python -> original_model = models.resnet50(pretrained=True).cuda().eval() -> torch.save(original_model, 'std_pretrained_resnet50_gpu.pt', _use_new_zipfile_serialization=False) -> ``` - -## Benchmark - -Take a look at the [Benchmark](docs/Benchmark.md). - -## Acknowledgement -Poros has been incubated for more than 2 years. In this project, NVIDIA helped us a lot (especially Gary Ji, Vincent Zhang, Jie Fang). They answered lots of technical questions about GPU and gave us many suggestions. Appreciate their great support. diff --git a/poros/cmake/FindTensorRT.cmake b/poros/cmake/FindTensorRT.cmake deleted file mode 100644 index 32a3eb57cfa..00000000000 --- a/poros/cmake/FindTensorRT.cmake +++ /dev/null @@ -1,87 +0,0 @@ -##################################### -## tensorrt specific configuration ## -##################################### - -set(_TensorRT_SEARCHES) - -if (DEFINED ENV{TENSORRT_ROOT}) - set(_TensorRT_SEARCH_ROOT PATHS $ENV{TENSORRT_ROOT} NO_DEFAULT_PATH) - list(APPEND _TensorRT_SEARCHES _TensorRT_SEARCH_ROOT) -endif () - -if (DEFINED ENV{TensorRT_INCLUDE_DIR}) - set(TensorRT_INCLUDE_DIR $ENV{TensorRT_INCLUDE_DIR}) -endif () - -if (DEFINED ENV{TensorRT_LIBRARY}) - set(TensorRT_LIBRARY $ENV{TensorRT_LIBRARY}) -endif () - -# appends some common paths -set(_TensorRT_SEARCH_NORMAL - PATHS "/usr/src/tensorrt/" # or custom tensorrt path - PATHS "/usr/local/tensorrt/" # or custom tensorrt path - PATHS "${PROJECT_SOURCE_DIR}/third_party/TensorRT/" # or custom tensorrt path - ) -list(APPEND _TensorRT_SEARCHES _TensorRT_SEARCH_NORMAL) - -# Include dir -foreach (search ${_TensorRT_SEARCHES}) - find_path(TensorRT_INCLUDE_DIR NAMES NvInfer.h NvInferPlugin.h ${${search}} PATH_SUFFIXES include) -endforeach () - -if (NOT TensorRT_LIBRARY) - foreach (search ${_TensorRT_SEARCHES}) - find_library(TensorRT_LIBRARY NAMES nvinfer ${${search}} PATH_SUFFIXES lib) - endforeach () -endif () - -if (NOT TensorRT_PLUGIN_LIBRARY) - foreach (search ${_TensorRT_SEARCHES}) - find_library(TensorRT_PLUGIN_LIBRARY NAMES nvinfer_plugin ${${search}} PATH_SUFFIXES lib) - endforeach () -endif () - - - -if (TensorRT_INCLUDE_DIR AND EXISTS "${TensorRT_INCLUDE_DIR}/NvInferVersion.h") - file(STRINGS "${TensorRT_INCLUDE_DIR}/NvInferVersion.h" TensorRT_MAJOR REGEX "^#define NV_TENSORRT_MAJOR [0-9]+.*$") - file(STRINGS "${TensorRT_INCLUDE_DIR}/NvInferVersion.h" TensorRT_MINOR REGEX "^#define NV_TENSORRT_MINOR [0-9]+.*$") - file(STRINGS "${TensorRT_INCLUDE_DIR}/NvInferVersion.h" TensorRT_PATCH REGEX "^#define NV_TENSORRT_PATCH [0-9]+.*$") - - string(REGEX REPLACE "^#define NV_TENSORRT_MAJOR ([0-9]+).*$" "\\1" TensorRT_VERSION_MAJOR "${TensorRT_MAJOR}") - string(REGEX REPLACE "^#define NV_TENSORRT_MINOR ([0-9]+).*$" "\\1" TensorRT_VERSION_MINOR "${TensorRT_MINOR}") - string(REGEX REPLACE "^#define NV_TENSORRT_PATCH ([0-9]+).*$" "\\1" TensorRT_VERSION_PATCH "${TensorRT_PATCH}") - set(TensorRT_VERSION_STRING "${TensorRT_VERSION_MAJOR}.${TensorRT_VERSION_MINOR}.${TensorRT_VERSION_PATCH}") -endif () - - -include(FindPackageHandleStandardArgs) -FIND_PACKAGE_HANDLE_STANDARD_ARGS(TensorRT REQUIRED_VARS TensorRT_LIBRARY TensorRT_PLUGIN_LIBRARY TensorRT_INCLUDE_DIR VERSION_VAR TensorRT_VERSION_STRING) -message(STATUS "TensorRT_LIBRARY: ${TensorRT_LIBRARY}") -message(STATUS "TensorRT_PLUGIN_LIBRARY: ${TensorRT_PLUGIN_LIBRARY}") -message(STATUS "TensorRT_INCLUDE_DIR: ${TensorRT_INCLUDE_DIR}") -message(STATUS "TensorRT: ${TensorRT_VERSION_STRING}") -if (TensorRT_FOUND) - set(TensorRT_INCLUDE_DIRS ${TensorRT_INCLUDE_DIR}) - - if (NOT TensorRT_LIBRARIES) - set(TensorRT_LIBRARIES ${TensorRT_LIBRARY}) - endif () - - if (NOT TARGET TensorRT::TensorRT) - add_library(TensorRT::TensorRT UNKNOWN IMPORTED) - set_target_properties(TensorRT::TensorRT PROPERTIES INTERFACE_INCLUDE_DIRECTORIES "${TensorRT_INCLUDE_DIRS}") - set_property(TARGET TensorRT::TensorRT APPEND PROPERTY IMPORTED_LOCATION "${TensorRT_LIBRARY}") - set_property(TARGET TensorRT::TensorRT APPEND PROPERTY VERSION "${TensorRT_VERSION_STRING}") - endif () - - if (NOT TARGET TensorRT::Plugin) - add_library(TensorRT::Plugin UNKNOWN IMPORTED) - set_target_properties(TensorRT::Plugin PROPERTIES INTERFACE_INCLUDE_DIRECTORIES "${TensorRT_INCLUDE_DIRS}") - set_property(TARGET TensorRT::Plugin APPEND PROPERTY IMPORTED_LOCATION "${TensorRT_PLUGIN_LIBRARY}") - set_property(TARGET TensorRT::Plugin APPEND PROPERTY VERSION "${TensorRT_VERSION_STRING}") - endif () -endif () - - diff --git a/poros/docs/Benchmark.md b/poros/docs/Benchmark.md deleted file mode 100644 index a0021a4b602..00000000000 --- a/poros/docs/Benchmark.md +++ /dev/null @@ -1,130 +0,0 @@ -# Benchmark -## Environment -This benchmark is tested on CentOS7 with GPU `A10` and CPU `Intel(R) Xeon(R) Platinum 8350C CPU @ 2.60GHz`, and its environment is as follows: -| Package | Version | -|----------|----------| -| CUDA | 11.3 | -| cuDNN | 8.3.2.44 | -| TensorRT | 8.4.1.5 | -| Python | 3.8.13 | -| PyTorch | 1.12.1 | - -## Performance -The following is the result of comparison between pytorch eager and poros, which measured by average latency time (ms) of model infering 1000 times. - -### 1. ResNet50 -Input shape: bx3x224x224 -| Batch size | PyTorch (ms) | Poros (ms) | -|------------|--------------|-------------| -| 1 | 6.17 | 1.70 | -| 2 | 6.02 | 2.41 | -| 4 | 6.33 | 3.23 | -| 8 | 8.55 | 4.75 | -| 16 | 16.22 | 7.82 | -| 32 | 32.09 | 14.00 | - -Model source: https://github.com/pytorch/vision/blob/main/torchvision/models/resnet.py - -### 2. VGG16 -Input shape: bx3x224x224 -| Batch size | PyTorch (ms) | Poros (ms) | -|------------|--------------|-------------| -| 1 | 3.20 | 2.71 | -| 2 | 4.97 | 3.78 | -| 4 | 8.20 | 6.09 | -| 8 | 14.64 | 10.20 | -| 16 | 27.47 | 19.17 | -| 32 | 53.09 | 36.47 | - -Model source: https://github.com/pytorch/vision/blob/main/torchvision/models/vgg.py - -### 3. MobileNetV2 -Input shape: bx3x224x224 -| Batch size | PyTorch (ms) | Poros (ms) | -|------------|--------------|-------------| -| 1 | 3.85 | 0.65 | -| 2 | 3.75 | 0.86 | -| 4 | 3.90 | 1.19 | -| 8 | 4.18 | 2.08 | -| 16 | 8.43 | 3.83 | -| 32 | 16.57 | 7.14 | - -Model source: https://github.com/tonylins/pytorch-mobilenet-v2/blob/master/MobileNetV2.py - -### 4. InceptionV3 -Input shape: bx3x224x224 -| Batch size | PyTorch (ms) | Poros (ms) | -|------------|--------------|-------------| -| 1 | 10.05 | 2.51 | -| 2 | 10.13 | 3.22 | -| 4 | 10.08 | 3.70 | -| 8 | 10.15 | 4.95 | -| 16 | 12.51 | 7.11 | -| 32 | 21.43 | 11.22 | - -Model source: https://github.com/pytorch/vision/blob/main/torchvision/models/inception.py - -### 5. Efficientnet_b0 -Input shape: bx3x224x224 -| Batch size | PyTorch (ms) | Poros (ms) | -|------------|--------------|-------------| -| 1 | 8.28 | 1.28 | -| 2 | 8.50 | 1.57 | -| 4 | 8.49 | 2.29 | -| 8 | 8.83 | 3.65 | -| 16 | 10.65 | 6.62 | -| 32 | 20.51 | 12.51 | - -Model source: https://github.com/rwightman/pytorch-image-models/blob/main/timm/models/efficientnet.py - -### 6. Bert-base-uncased -Input shape: bx128 -| Batch size | PyTorch (ms) | Poros (ms) | -|------------|--------------|-------------| -| 1 | 6.40 | 2.02 | -| 2 | 7.14 | 2.59 | -| 4 | 11.58 | 4.39 | -| 8 | 21.64 | 8.41 | -| 16 | 44.20 | 16.90 | -| 32 | 92.69 | 32.21 | - -Model source: https://github.com/huggingface/transformers/blob/main/src/transformers/models/bert/modeling_bert.py - -### 7. Vision Transformer (ViT) -Input shape: bx3x224x224 -| Batch size | PyTorch (ms) | Poros (ms) | -|------------|--------------|-------------| -| 1 | 6.38 | 3.07 | -| 2 | 10.35 | 4.57 | -| 4 | 19.06 | 8.37 | -| 8 | 36.71 | 16.34 | -| 16 | 73.84 | 29.92 | -| 32 | 147.70 | 58.11 | - -Model source: https://github.com/rwightman/pytorch-image-models/blob/main/timm/models/vision_transformer.py - -### 8. YOLOv5s -Input shape: bx3x640x640 -| Batch size | PyTorch (ms) | Poros (ms) | -|------------|--------------|-------------| -| 1 | 6.17 | 2.22 | -| 2 | 5.93 | 3.96 | -| 4 | 10.02 | 6.84 | -| 8 | 20.02 | 12.86 | -| 16 | 38.17 | 24.80 | -| 32 | 77.19 | 49.16 | - -Model source: https://github.com/ultralytics/yolov5/blob/master/models/yolo.py - -### 9. Swin Transformer -Input shape: bx3x224x224 -| Batch size | PyTorch (ms) | Poros (ms) | -|------------|--------------|-------------| -| 1 | 14.11 | 7.68 | -| 2 | 22.73 | 11.99 | -| 4 | 42.21 | 21.74 | -| 8 | 83.07 | 42.18 | -| 16 | 162.34 | 78.34 | -| 32 | 317.43 | 149.72 | - -Model source: https://github.com/microsoft/Swin-Transformer/blob/main/models/swin_transformer.py \ No newline at end of file diff --git a/poros/example/test_resnet.py b/poros/example/test_resnet.py deleted file mode 100644 index 7c9ad1f8780..00000000000 --- a/poros/example/test_resnet.py +++ /dev/null @@ -1,90 +0,0 @@ -# Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -""" -test resnet50 -""" - -import time -import poros -import torch -from torchvision import models - -torch.set_grad_enabled(False) - -def load_example_input_datas(): - """fake data""" - data_list = [] - input_1 = torch.randn(1, 3, 224, 224, dtype=torch.float32).cuda() - data_list.append(input_1) - return data_list - - -if __name__ == '__main__': - - input_datas = load_example_input_datas() - original_model = models.resnet50(pretrained=True).cuda().eval() - - option = poros.PorosOptions() - # option.max_workspace_size = 1 << 30 - # option.is_dynamic = False - # option.debug = True - # option.unconst_ops_thres = 0 - - - try: - poros_model = poros.compile(torch.jit.script(original_model), input_datas, option) - except Exception as e: - print("compile poros_model failed. error msg: {}".format(e)) - exit(0) - - - for input in input_datas: - ori_res = original_model(input) - poros_res = poros_model(input) - res_diff = torch.abs(ori_res - poros_res) - print("max_diff", torch.max(res_diff)) - print(poros_res.shape) - - # warm up - for i in range (100): - for input in input_datas: - ori_res = original_model(input) - poros_res = poros_model(input) - - count = 1000 - - # POROS benchmark - torch.cuda.synchronize() - st = time.time() - for i in range (count): - # step4: 预测。 - for input in input_datas: - poros_res = poros_model(input) - - torch.cuda.synchronize() - poros_elapsed_time = time.time() - st - print("poros infer time:{:.5f}ms/infer".format(poros_elapsed_time)) - - # original benchmark - torch.cuda.synchronize() - st = time.time() - for i in range (count): - # step4: 预测。 - for input in input_datas: - ori_res = original_model(input) - - torch.cuda.synchronize() - original_elapsed_time = time.time() - st - print("original infer time/:{:.5f}ms/infer".format(original_elapsed_time)) - print("speedup: +{:.2f}%".format((original_elapsed_time / poros_elapsed_time - 1 ) * 100)) \ No newline at end of file diff --git a/poros/poros/compile/compile.cpp b/poros/poros/compile/compile.cpp deleted file mode 100644 index 9f666988750..00000000000 --- a/poros/poros/compile/compile.cpp +++ /dev/null @@ -1,482 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file compile.h -* @author tianjinjin@baidu.com -* @author huangben@baidu.com -* @date Fri Mar 5 11:39:03 CST 2021 -* @brief -**/ -#include "poros/compile/compile.h" - -//pytorch -#include - -//pytorch passes -#include -#include -#include -#include -#include -//#includ -#include -#include -#include -#include -#include -#include -#include -#include -#include - -#include "poros/compile/graph_prewarm.h" -#include "poros/compile/graph_segment.h" -#include "poros/context/poros_global.h" -#include "poros/lowering/lowering_pass.h" -#include "poros/lowering/op_fuse_pass.h" -#include "poros/lowering/segment_post_processing.h" -#include "poros/util/poros_util.h" -// #include "poros/iplugin/plugin_create.h" - -namespace baidu { -namespace mirana { -namespace poros { - -Compiler::~Compiler() { - close(); -} - -int Compiler::init(const PorosOptions& options) { - _options = options; - if (options.debug == true) { - // when setting this, all the INFO level will be printed - c10::ShowLogInfoToStderr(); - } - if (_options.unconst_ops_thres == -1) { - _options.unconst_ops_thres = 10; - if (_options.device == Device::XPU) { - _options.unconst_ops_thres = 1; - } - } - PorosGlobalContext::instance().set_poros_options(_options); - return 0; -} - -int Compiler::preprocess_graph(std::shared_ptr& graph) { - GRAPH_DEBUG("Before preprocess_graph:", graph); - - //step1: some passes provided by pytorch that insensitive for profile - { - torch::jit::Inline(*graph); - GRAPH_DUMP("inline graph:", graph); - /*PropagateInputShapes and PropagateRequiresGrad maybe no need anymore*/ - torch::jit::PropagateInputShapes(graph); - - torch::jit::ClearProfilingInformation(graph); - torch::jit::LowerGradOf(*graph); //TODO: maybe no need - torch::jit::PropagateRequiresGrad(graph); - - torch::jit::runRequiredPasses(graph); - /* DecomposeOps try to replace addmm & batch_norm & layer_norm. - considering poros can handle batch_norm. so we do not unfold this. */ - //torch::jit::DecomposeOps(graph); - torch::jit::ConstantPropagation(graph); - torch::jit::EliminateDeadCode(graph); - torch::jit::EliminateCommonSubexpression(graph); - torch::jit::ConstantPooling(graph); - torch::jit::PeepholeOptimize(graph, false); - torch::jit::EliminateDeadCode(graph); - torch::jit::LowerSimpleTuples(graph); // TODO: maybe should not be here - torch::jit::CheckInplace(graph); - /*DO NOT set LowerAllTuples. - * this may lead method output is not tuple. - * and get error message like this: "Method (but not graphs in general) require a single output." */ - // torch::jit::LowerAllTuples(graph); - } - - //step2: some passes provided by poros that insensitive for profile - { - baidu::mirana::poros::replace_illegal_constant(graph); - baidu::mirana::poros::eliminate_exception_pass(graph); - baidu::mirana::poros::eliminate_maxpool_with_indices(graph); - baidu::mirana::poros::eliminate_simple_useless_nodes(graph); - baidu::mirana::poros::unpack_std(graph); - baidu::mirana::poros::unpack_var(graph); - baidu::mirana::poros::replace_log_softmax(graph); - baidu::mirana::poros::replace_log_sigmoid(graph); - baidu::mirana::poros::replace_pad(graph); - baidu::mirana::poros::fuse_ops_preprocess(graph); - torch::jit::runRequiredPasses(graph); - } - - GRAPH_DEBUG("After preprocess_graph:", graph); - return 0; -} - -int Compiler::compile(const torch::jit::Module& origin_module, - const ivalue_vec_t& prewarm_datas, torch::jit::Module* optimized_module) { - - _origin_module = &origin_module; - _prewarm_datas = prewarm_datas; - - GRAPH_DUMP("origin_module graph:", origin_module.get_method("forward").graph()); - torch::jit::setGraphExecutorOptimize(true); - - std::shared_ptr opt_graph = nullptr; - { - //step1: clone orign module to unfold module - torch::jit::Module intermediate_module = torch::jit::freeze_module(origin_module); - auto method = intermediate_module.get_method("forward"); - auto graph = method.graph(); - int ret = preprocess_graph(graph); - if (ret < 0) { - LOG(ERROR) << "preprocess_graph failed!"; - return -1; - } - //attention. graph copy happened inside LowerGraph function - auto graph_and_ivalues = torch::jit::LowerGraph(*graph, intermediate_module._ivalue()); - opt_graph = graph_and_ivalues.first; - } - - std::shared_ptr prewarm_graph = graph_prewarm(opt_graph, prewarm_datas); - GRAPH_DUMP("prewarmed_module graph:", prewarm_graph); - - //cpu的话,预热后就返回 - if (_options.device == Device::CPU) { - merge_graph_to_module(prewarm_graph, *optimized_module, true); - return 0; - } - - //step2: try to find segments in unfold module - //划分完子图的模型 - int ret = segment_graph(prewarm_graph); - if (ret < 0) { - LOG(ERROR) << "segment_graph failed!"; - return -1; - } - GRAPH_DUMP("segmented_module graph:", prewarm_graph); - - //step3: try to replace subgraph to engine graph - merge_graph_to_module(prewarm_graph, *optimized_module, true); - ret = optimize_subgraph(prewarm_graph, optimized_module); - if (ret < 0) { - LOG(ERROR) << "optimize_subgraph failed!"; - return -1; - } - GRAPH_DUMP("optimized_module graph:", optimized_module->get_method("forward").graph()); - return 0; -} - -int Compiler::segment_graph(std::shared_ptr& g) { - - IEngine* engine(nullptr); - std::string engine_name(""); - if (_options.device == Device::CUDA) { - engine_name = "TensorrtEngine"; - } else if (_options.device == Device::XPU) { - engine_name = "XtclEngine"; - } else { - engine = nullptr; - } - - if (engine_name != "") { - engine = dynamic_cast(create_plugin(engine_name, - PorosGlobalContext::instance()._engine_creator_map)); - if (engine->init() < 0) { - delete engine; - return -1; - } - } - - graph_segment(g, engine); - GRAPH_DEBUG("After segment graph:", g); - delete engine; - return 0; -} - -IEngine* Compiler::select_engine(const torch::jit::Node* n) { - if (n == nullptr || n->kind() != torch::jit::prim::CudaFusionGroup) { - return nullptr; - } - - IEngine* engine(nullptr); - std::string engine_name(""); - if (_options.device == Device::CUDA) { - engine_name = "TensorrtEngine"; - } else if (_options.device == Device::XPU) { - engine_name = "XtclEngine"; - } else { - engine = nullptr; - } - - if (engine_name != "") { - engine = dynamic_cast(create_plugin(engine_name, - PorosGlobalContext::instance()._engine_creator_map)); - if (engine->init() < 0) { - return nullptr; - } - _engine_map[n] = engine; - } - - return engine; -} - - -int Compiler::optimize_subgraph(const std::shared_ptr& opt_graph, - torch::jit::Module* optimized_module) { - auto block = opt_graph->block(); - auto ret = optimize_subblock(block, optimized_module); - return ret; -} - - -int Compiler::optimize_subblock(torch::jit::Block* block, - torch::jit::Module* optimized_module) { - - std::vector to_optimized_nodes; - // 避免有的cudafusiongroup在子block内,需要这样遍历 - find_to_optimized_nodes(block, to_optimized_nodes); - //保险起见,再sort一遍。 - std::sort(to_optimized_nodes.begin(), to_optimized_nodes.end(), [&](torch::jit::Node* a, torch::jit::Node* b) { - return a->isBefore(b); - }); - - //size_t i = to_optimized_nodes.size(); - for (auto iter = to_optimized_nodes.rbegin(); iter != to_optimized_nodes.rend(); iter++) { - auto node = *iter; - - // todo: - // 1、目前不支持scalar的输出。若强行转engine输出的是tensor,与后面scalar类型不匹配,整个模型跑不起来。 - // 2、subgraph没有输入的情况。 - // 遇到这两种情况先unmerge掉,输出scalar的情况待支持 06.20 - if (node->inputs().size() == 0) { - LOG(WARNING) << "Subgraph: " << node_info_with_attr(node) << " has no input. unmerge it."; - baidu::mirana::poros::unmerge_subgraph(node); - continue; - } - bool node_should_be_unmerged = false; - for (size_t i = 0; i < node->outputs().size(); i++) { - if (node->outputs()[i]->type()->kind() != c10::TypeKind::TensorType && - !node->outputs()[i]->type()->isSubtypeOf(c10::ListType::ofTensors())) { - LOG(WARNING) << "Subgraph: " << node_info_with_attr(node) << " outputs contain non-tensor or non-tensor[] values. unmerge it."; - node_should_be_unmerged = true; - baidu::mirana::poros::unmerge_subgraph(node); - break; - } - } - - if (node_should_be_unmerged) { - continue; - } - - IEngine* engine = select_engine(node); - if (engine == nullptr) { - LOG(ERROR) << "can't find Engine for node: " << node->kind().toQualString(); - return -1; - } - std::shared_ptr subgraph = node->g(torch::jit::attr::Subgraph); - LOG(INFO) << "\n \n ###########\n \n" - << "start to optimize graph: " << node_info_with_attr(node); - - int non_constant_node_num = 0; - auto subblock = subgraph->block(); - - //engine->transform(*node, *optimized_module); - for (auto it = subblock->nodes().begin(); it != subblock->nodes().end(); ++it) { - if (it->kind() != torch::jit::prim::Constant) { - non_constant_node_num ++; - if (non_constant_node_num > _options.unconst_ops_thres) { - break; - } - } - } - if (non_constant_node_num <= _options.unconst_ops_thres) { - LOG(INFO) << "subgraph size is too small, unmerge it."; - baidu::mirana::poros::unmerge_subgraph(node); - } else { - if (transform(engine, *node, *optimized_module) < 0) { - LOG(WARNING) << "transform failed, use origin sub_graph"; - if (_options.debug) { - GRAPH_DUMP("transform failed graph: ", subgraph); - return -1; - } - } - } - } - - for (auto it = block->nodes().begin(); it != block->nodes().end(); it++) { - for (torch::jit::Block* ib : it->blocks()) { - optimize_subblock(ib, optimized_module); - } - } - return 0; -} - -void Compiler::close() { - - for (auto&e : _engine_map) { - delete e.second; - } -} - -int Compiler::transform(IEngine* engine, torch::jit::Node& subgraph_node, - torch::jit::Module& module) { - - AT_ASSERT(subgraph_node.kind() == torch::jit::prim::CudaFusionGroup); - std::shared_ptr sub_graph_copy = subgraph_node.g(torch::jit::attr::Subgraph)->copy(); - - std::string serialized_engine; - - // 在拷贝的子图上删除无用的节点 - if (!eliminate_subgraph_useless_nodes(sub_graph_copy, subgraph_node, false)) { - baidu::mirana::poros::unmerge_subgraph(&subgraph_node); - return -1; - } - - PorosGraph poros_graph = {sub_graph_copy.get(), &subgraph_node}; - - //++poros_graph.allocated_index; - //int ret = engine->transform(poros_graph, serialized_engine); - int ret = engine->transform(poros_graph); - if (ret < 0) { - baidu::mirana::poros::unmerge_subgraph(&subgraph_node); - return ret; - } - // 子图转换成功,删除(替换)子图输入无用的节点 - std::shared_ptr sub_graph = subgraph_node.g(torch::jit::attr::Subgraph); - eliminate_subgraph_useless_nodes(sub_graph, subgraph_node, true); - - auto parent_graph = subgraph_node.owningGraph(); - - // 有的子图输出的tensor原本是long类型,但是trtengine只支持int类型。 - // 那么就需要在engine后添加aten::to(long)的操作将其还原回去。 - // 避免有的op会强制检查long类型(例如:aten::index) - subgraph_outputs_int2long(parent_graph, subgraph_node, sub_graph_copy); - - - //std::pair num_io = engine_ptr->num_io; - //AT_ASSERT(num_io.first == sub_graph->inputs().size()); - //AT_ASSERT(num_io.second == sub_graph->outputs().size()); - - //add engine to attribute - std::string name = engine->who_am_i() + "_" + std::to_string(_engine_index++); - engine->register_module_attribute(name, module); - - //get self input. it's about the module func - torch::jit::Value* self = nullptr; - auto first_input_c = parent_graph->inputs()[0]->type()->cast(); - if (first_input_c->is_module()) { - self = parent_graph->inputs()[0]; - } else { - self = parent_graph->insertInput(0, "self"); //should set as the first input param - self->setType(module._ivalue()->type()); - } - - torch::jit::WithInsertPoint guard(&subgraph_node); - //build new node & remove old graph - auto engine_node = parent_graph->createGetAttr(self, name); - engine_node->insertBefore(&subgraph_node); - - std::vector engine_inputs; - for (auto input : subgraph_node.inputs()) { - //TODO: consider situation that when input is not a tensor - engine_inputs.push_back(input); - } - - auto input_list_node = parent_graph->createList(c10::TensorType::get(), - torch::jit::ArrayRef(engine_inputs)); - input_list_node->insertBefore(&subgraph_node); - - std::vector execute_node_inputs; - execute_node_inputs.push_back(input_list_node->outputs()[0]); - execute_node_inputs.push_back(engine_node->outputs()[0]); - - auto execute_node = parent_graph->create( - c10::Symbol::fromQualString(engine->who_am_i() + "::execute_engine"), - torch::jit::ArrayRef(execute_node_inputs), - 1); - execute_node->insertBefore(&subgraph_node); - execute_node->outputs()[0]->setType(c10::ListType::ofTensors()); - - //auto unpack_node = parent_graph->createListUnpack(execute_node->outputs()[0], num_io.second); - auto unpack_node = parent_graph->createListUnpack(execute_node->outputs()[0], subgraph_node.outputs().size()); - unpack_node->insertBefore(&subgraph_node); - - //AT_ASSERT(subgraph_node.outputs().size() == unpack_node->outputs().size()); - for (size_t idx = 0; idx < unpack_node->outputs().size(); idx++) { - subgraph_node.outputs()[idx]->replaceAllUsesWith(unpack_node->outputs()[idx]); - } - - subgraph_node.removeAllInputs(); - subgraph_node.destroy(); //TODO: 没有清理subgraph,可能会有内存泄漏,确认一下 - return 0; -} - -/** - * @brief compile graph - * - * @param [in] module : 原始module - * @param [in] input_ivalues : 预热数据 - * @param [in] options : 参数 - * @return optimized_module - * @retval !nullptr => succeed nullptr => failed - **/ -std::unique_ptr CompileGraph(const torch::jit::Module& origin_module, - const std::vector >& prewarm_datas, - const PorosOptions& options) { - Compiler compiler; - if (compiler.init(options) < 0) { - return nullptr; - } - - try { - std::unique_ptr optimized_module(new torch::jit::Module(origin_module._ivalue()->name() + "_poros")); - if (compiler.compile(origin_module, prewarm_datas, optimized_module.get()) < 0) { - return nullptr; - } - - return optimized_module; - } catch (const c10::Error& e) { - LOG(ERROR) << e.msg(); - return nullptr; - } -} - -std::unique_ptr Compile(const torch::jit::Module& module, - const std::vector >& prewarm_datas, - const PorosOptions& options) { - - auto compiled_module = CompileGraph(module, prewarm_datas, options); - if (compiled_module) { - std::unique_ptr poros_module(new PorosModule(*compiled_module)); - poros_module->_options = options; - - if (options.device == Device::CUDA) { - poros_module->to(at::kCUDA); - } - - if (options.debug == true) { - // when setting this, all the INFO level will be printed - c10::ShowLogInfoToStderr(); - } - return poros_module; - } else { - return nullptr; - } -} - -}//poros -}//mirana -}//baidu diff --git a/poros/poros/compile/compile.h b/poros/poros/compile/compile.h deleted file mode 100644 index 702a981769f..00000000000 --- a/poros/poros/compile/compile.h +++ /dev/null @@ -1,169 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file compile.h -* @author tianjinjin@baidu.com -* @author huangben@baidu.com -* @date Fri Mar 5 11:39:03 CST 2021 -* @brief -**/ - -#pragma once - -#include -#include -#include -#include -#include - -#include "poros/engine/iengine.h" -#include "poros/compile/poros_module.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/** - * @brief compile graph - * - * @param [in] module : 原始module - * @param [in] input_ivalues : 预热数据 - * @param [in] options : 参数 - * @return porosmodule - * @retval !nullptr => succeed nullptr => failed - **/ -std::unique_ptr Compile(const torch::jit::Module& module, - const std::vector >& prewarm_datas, - const PorosOptions& options); - -class Compiler { -public: - typedef std::unordered_map engine_map_t; - typedef std::vector > ivalue_vec_t; - - Compiler() : _origin_module(NULL) {} - ~Compiler(); - - /** - * @brief initial Compiler - * - * @param [in] options : poros options - * @return int - * @retval 0 => succeed <0 => failed - **/ - int init(const PorosOptions& options); - - /** - * @brief compile whole graph - * - * @param [in] origin_module - * @param [in] prewarm_datas : ivalue_vec_t, vector of IValue - * @param [out] optimized_module : optimized graph - * @return int - * @retval 0 => succeed <0 => failed - **/ - int compile(const torch::jit::Module& origin_module, - const ivalue_vec_t& prewarm_datas, - torch::jit::Module* optimized_module); - -private: - - /** - * @brief preprocess this calculation graph - * - * @param [out] graph : preprcessed graph - * @return int - * @retval 0 => succeed <0 => failed - **/ - int preprocess_graph(std::shared_ptr& graph); - - /** - * @brief segement this calculation graph - * - * @param [in/out] graph - * @return int - * @retval 0 => succeed <0 => failed - **/ - int segment_graph(std::shared_ptr& graph); - - - //子图优化 - /** - * @brief optimize this calculation graph - * - * @param [in] opt_graph : - * @param [out] optimized_module : optimized module - * @return int - * @retval 0 => succeed <0 => failed - **/ - int optimize_subgraph(const std::shared_ptr& opt_graph, - torch::jit::Module* optimized_module); - - //子图优化(block) - int optimize_subblock(torch::jit::Block* block, - torch::jit::Module* optimized_module); - - /** - * @brief 将子图基于engine编译成新图 - * - * @param [in] engine : 子图用到的engine - * @param [in] subgraph_node : 子图结点 - * @return [out] module : 转化后的模型 - * @retval 0 => succeed <0 => failed - **/ - int transform(IEngine* engine, torch::jit::Node& subgraph_node, - torch::jit::Module& module); - - /** - * @brief 根据子图和options选择engine - * - * @param [in] node : 子图代表结点 - * @return int - * @retval 0 => succeed <0 => failed - **/ - IEngine* select_engine(const torch::jit::Node* n); - - /** - * @brief destory - * - * @return void - **/ - void close(); - -private: - int _max_segment_depth{5}; //最大子图分割深度 - ivalue_vec_t _prewarm_datas; //预热用的输入数据 - PorosOptions _options; - engine_map_t _engine_map; //记录子图用的engine - const torch::jit::Module* _origin_module; //原始模型 - std::atomic _engine_index = {0}; //记录engine的index -}; - -/** - * @brief compile graph, 内部使用 - * - * @param [in] module : 原始module - * @param [in] input_ivalues : 预热数据 - * @param [in] options : 参数 - * @return optimized_module - * @retval !nullptr => succeed nullptr => failed - **/ -std::unique_ptr CompileGraph(const torch::jit::Module& module, - const std::vector >& prewarm_datas, - const PorosOptions& options); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/compile/graph_prewarm.cpp b/poros/poros/compile/graph_prewarm.cpp deleted file mode 100644 index 645beea9c4c..00000000000 --- a/poros/poros/compile/graph_prewarm.cpp +++ /dev/null @@ -1,191 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file graph_prewarm.cpp -* @author tianjinjin@baidu.com -* @date Fri Apr 23 11:41:59 CST 2021 -* @brief -**/ -#include "poros/compile/graph_prewarm.h" - -//pytorch -#include -#include -#include -#include -#include -#include -// #include -#include -#include -#include -#include - -#include "poros/compile/ivalues_analysis.h" -#include "poros/lowering/lowering_pass.h" -#include "poros/lowering/op_fuse_pass.h" - -namespace baidu { -namespace mirana { -namespace poros { - -namespace { -using namespace torch::jit; -using Stack = std::vector; - -struct PorosGraphPrewarm { - PorosGraphPrewarm(std::shared_ptr graph):graph_(std::move(graph)) {} - - //graph 预热核心逻辑 - std::shared_ptr run(std::vector& stack_vec) { - - //step1: back up the prewarm data - //attention: the data in stack has changed during the interpreterstate procedure. - //so we should copy twice the stack before interpreter execution - //one for input_param_propagate, and the other for final interpreterState. - std::vector stack_vec_final = stack_vec; - std::vector stack_vec_copy = stack_vec; - - //step2: the first round ivalue analysis - torch::jit::getProfilingMode() = true; - torch::jit::getExecutorMode() = true; - torch::jit::setGraphExecutorOptimize(false); - torch::jit::getNumProfiledRuns() = stack_vec.size(); - GRAPH_DEBUG("before first round IvalueAnalysis Graph: ", graph_); - std::unique_ptr ia = IvalueAnalysis::analysis_ivalue_for_graph(graph_); - ExecutionPlan plan = ExecutionPlan(ia->graph(), "first_round_prewarm"); - for (size_t i = 0; i < stack_vec.size(); i++) { - InterpreterState(plan.code).run(stack_vec[i]); - } - std::shared_ptr output_graph = ia->graph(); - GRAPH_DEBUG("after first round IvalueAnalysis Graph: ", output_graph); - - //step3: necessary passes to eliminate the profile information in graph - { - baidu::mirana::poros::input_param_propagate(output_graph, stack_vec_copy); - std::vector().swap(stack_vec_copy); - GRAPH_DEBUG("after input_param_propagate Graph: ", output_graph); - torch::jit::ProfilingRecord::removeProfileCounter(output_graph->block()); - GRAPH_DEBUG("after removeProfileCounter Graph: ", output_graph); - baidu::mirana::poros::remove_simple_type_profile_nodes(output_graph); - GRAPH_DEBUG("after remove_simple_type_profile_nodes Graph: ", output_graph); - torch::jit::RemoveProfileNodesAndSpecializeTypes(output_graph); - GRAPH_DEBUG("after RemoveProfileNodesAndSpecializeTypes Graph: ", output_graph); - } - - // step4: some passes can be run based on the prifiled graph and data - { - torch::jit::runRequiredPasses(output_graph); - torch::jit::EliminateDeadCode(output_graph); - torch::jit::EliminateCommonSubexpression(output_graph); - /* addmm is only done as an optimization for onnx, so we disable it */ - torch::jit::PeepholeOptimize(output_graph, /*addmm_fusion_enabled*/false); - torch::jit::ConstantPropagation(output_graph); //this is very necessary for prone if block!!!! - torch::jit::ConstantPooling(output_graph); - // torch::jit::UnrollLoops(output_graph); - baidu::mirana::poros::unrolling_loop(output_graph); - baidu::mirana::poros::freeze_percentformat(output_graph); - baidu::mirana::poros::freeze_aten_size(output_graph); - baidu::mirana::poros::freeze_aten_len(output_graph); - baidu::mirana::poros::unrolling_loop(output_graph); - - //some mutation handle pass below - GRAPH_DEBUG("before remove mutation Graph: ", output_graph); - //TODO: handle this later, it cores when we using prepare_inplace_ops. - //prepare_inplace_ops(output_graph); - torch::jit::RemoveListMutation(output_graph); - torch::jit::RemoveTensorMutation(output_graph); - GRAPH_DEBUG("after remove mutation Graph: ", output_graph); - - baidu::mirana::poros::fuse_ops_prewarm(output_graph); - - // run some pass again after unrolled loops - torch::jit::PeepholeOptimize(output_graph, /*addmm_fusion_enabled*/false); - torch::jit::ConstantPropagation(output_graph); - torch::jit::EliminateCommonSubexpression(output_graph); - torch::jit::CheckInplace(output_graph); - torch::jit::runRequiredPasses(output_graph); - - baidu::mirana::poros::eliminate_some_dict(output_graph); - baidu::mirana::poros::eliminate_some_list(output_graph); - - torch::jit::PeepholeOptimize(output_graph, /*addmm_fusion_enabled*/false); - torch::jit::ConstantPropagation(output_graph); - torch::jit::ConstantPooling(output_graph); - baidu::mirana::poros::unrolling_loop(output_graph); - torch::jit::EliminateCommonSubexpression(output_graph); - baidu::mirana::poros::link_mutable_list(output_graph); - torch::jit::CheckInplace(output_graph); - torch::jit::runRequiredPasses(output_graph); - - torch::jit::LowerSimpleTuples(output_graph); - } - - //step5: prepare for second round ivalue analysis - //reset the profile number & clean profile information last round. - torch::jit::getNumProfiledRuns() = stack_vec_final.size(); - torch::jit::ClearProfilingInformation(output_graph); - std::unique_ptr ia_final = IvalueAnalysis::analysis_ivalue_for_graph(output_graph); - ExecutionPlan plan_final = ExecutionPlan(ia_final->graph(), "second_round_prewarm"); - for (size_t i = 0; i < stack_vec_final.size(); i++) { - InterpreterState(plan_final.code).run(stack_vec_final[i]); - } - std::shared_ptr final_graph = ia_final->graph(); - - //step6: store the final dynamic information - bool is_dynamic_shape = PorosGlobalContext::instance().get_poros_options().is_dynamic; - if (is_dynamic_shape) { - ia_final->gen_value_dyanamic_shape(); - } - ia_final->gen_list_size(); - ia_final->gen_int_intlist_value(); - - //step7: necessary passes to eliminate the profile record - { - torch::jit::ProfilingRecord::removeProfileCounter(final_graph->block()); - baidu::mirana::poros::remove_simple_type_profile_nodes(final_graph); - torch::jit::RemoveProfileNodesAndSpecializeTypes(final_graph); - baidu::mirana::poros::freeze_aten_dim(final_graph); - baidu::mirana::poros::freeze_list_construct(final_graph); - } - - GRAPH_DUMP("final graph_prewarm Graph: ", final_graph); - return final_graph; - } - -private: - std::shared_ptr graph_; - -}; // struct PorosGraphPrewarm - -} // anonymous namespace - -std::shared_ptr graph_prewarm(std::shared_ptr& graph, - const std::vector >& prewarm_datas) { - std::vector> stacks; - for (size_t i = 0; i < prewarm_datas.size(); ++i) { //TODO: Make it better here - std::vector stack; - for (c10::IValue input : prewarm_datas[i]) { - stack.push_back(input); - } - stacks.push_back(stack); - } - std::shared_ptr new_graph = PorosGraphPrewarm(graph).run(stacks); - return new_graph; -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/compile/graph_prewarm.h b/poros/poros/compile/graph_prewarm.h deleted file mode 100644 index 417eaabcd12..00000000000 --- a/poros/poros/compile/graph_prewarm.h +++ /dev/null @@ -1,55 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file graph_prewarm.h -* @author tianjinjin@baidu.com -* @date Thu Mar 18 14:33:54 CST 2021 -* @brief -**/ - -#pragma once - -//pytorch -#include - -namespace baidu { -namespace mirana { -namespace poros { - -/** - * @brief 首先,graph_prewarm 将给定的预热数据‘prewarm_datas’喂给添加了profile节点的graph。 - * 从而得到graph中每个节点的我们期望存储的信息(tensor类节点的dim信息/ bool类节点的值 等..) - * 其次,graph_prewarm 基于这份存储的信息,进一步对graph进行图层面的优化, - * 最后,graph_prewarm 返回一份完成了图优化的graph。 - * - * first, graph_prewarm feed the given prewarm_datas to the graph which has added many profile nodes. - * we can get much information (dim information of tensor value / exact numerical value of - * the bool value, etc...) about each node in the graph by profiling the graph。 - * - * second,graph_prewarm optimize the given graph at the graph level based on the profile data we captured - * - * finally,graph_prewarm return the graph that has fully optimized in graph level - * - * @param [in] graph : the graph to be warmed - * @param [in] prewarm_datas : prewarm data - * @return prewarmed_graph - **/ -std::shared_ptr graph_prewarm( - std::shared_ptr& graph, - const std::vector >& prewarm_datas); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/compile/graph_segment.cpp b/poros/poros/compile/graph_segment.cpp deleted file mode 100644 index 61b09bb6b8d..00000000000 --- a/poros/poros/compile/graph_segment.cpp +++ /dev/null @@ -1,1308 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file graph_segment.cpp -* @author tianjinjin@baidu.com -* @author tianshaoqing@baidu.com -* @date Fri Mar 19 19:18:20 CST 2021 -* @brief -**/ -#include "poros/compile/graph_segment.h" - -//pytorch -#include -#include -#include -#include -#include -#include -// #include - -#include "poros/context/poros_global.h" -#include "poros/lowering/lowering_pass.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -namespace { -using namespace torch::jit; - -Value* broadcast_sizes(at::ArrayRef sizes) { - AT_ASSERT(!sizes.empty()); - Graph* graph = sizes[0]->owningGraph(); - Node* broadcast_n = - graph->insertNode(graph->create(prim::BroadcastSizes, sizes)); - broadcast_n->output()->setType(ListType::ofInts()); - return broadcast_n->output(); -} - -struct PorosGraphSegment { - using FusionCallback = std::function; - Block* block_; - std::unique_ptr aliasDb_; - std::shared_ptr graph_; - Symbol kind_ = prim::CudaFusionGroup; - IEngine* engine_; - - PorosGraphSegment(Block* block, std::shared_ptr graph, IEngine* engine) - : block_(block), graph_(std::move(graph)), engine_(engine) {} - - //判断一个节点和它某个输入的关系,是否该输入的所有消费者已经在这group里了。 - bool all_users_are_this_cunsumer(Node* consumer, Value* producer) { - Node* defining_node = producer->node(); - for (Value* o : defining_node->outputs()) { - for (auto u : o->uses()) { - if (u.user != consumer && - !(u.user->matches("aten::size(Tensor self) -> int[]"))) { - return false; - } - } - } - return true; - } - - //判断给定的一个节点(node),当前engine是否支持。 - bool is_node_fusable(const Node* node) { - //针对aten::append需要额外的判断条件。 - //当aten::append位于某个block,而它可能改变其parentblock中的ListConstruct的产出的时候,整个逻辑就gg了。 - //本质上,这还是inplace 语义导致的问题。 - //poros需要针对 inplace语义的算子,做更好的预处理逻辑。 - // if ((node->kind() == aten::append) && engine_->is_node_supported(node) && - // node->inputs().at(0)->node()->kind() == prim::ListConstruct) { - // const Node* mutable_node = node->inputs().at(0)->node(); - // for (auto &use: mutable_node->output()->uses()) { - // if (use.user->owningBlock() != mutable_node->owningBlock()) { - // LOG(WARNING) << "opps! meet mutable aten::append: " << node_info(node); - // return false; - // } - // } - // return true; - // } - // loop和if子block里的mutable op还没有串联,fuse会有问题,先禁用 04.22 - if (PorosGlobalContext::instance().supported_mutable_ops_set.count(node->kind()) > 0 - && engine_->is_node_supported(node)) { - if (node->owningBlock() != node->owningGraph()->block()) { - LOG(WARNING) << "Graph fuser meets mutable op in sub_block, which is not" - " yet supported. Node info: " << node_info(node); - return false; - } - } - - // aten::__getitem__ idx参数不支持非constant类型 - if (node->kind() == torch::jit::aten::__getitem__) { - if (node->inputs().size() == 2 && - node->input(1)->node()->kind() != torch::jit::prim::Constant) { - LOG(WARNING) << "The index input of aten::__getitem__ is not supported as non-constant type."; - return false; - } - } - - // aten::_set_item idx参数不支持非constant类型 - if (node->kind() == torch::jit::aten::_set_item) { - if (node->inputs().size() == 3 && - node->input(1)->node()->kind() != torch::jit::prim::Constant) { - LOG(WARNING) << "The index input of aten::_set_item is not supported as non-constant type."; - return false; - } - } - - if (node->kind() == kind_ || engine_->is_node_supported(node)) { - return true; - } - return false; - } - - //TODO: this should be better - //判断给定的一个节点(node),是否可以fuse到已有的节点组(fusion)里面去。 - bool is_node_fusable(const Node* fusion, const Node* node) { - //对prim::ListConstruct 这种有引用语义的,需要更严格的校验。 - // if (node->kind() == prim::ListConstruct && is_node_fusable(node)) { - // for (auto &use: node->output()->uses()) { - // if (use.user->owningBlock() != fusion->owningBlock()) { - // LOG(WARNING) << "opps! meet mutable ListConstruct: " << node_info(node); - // return false; - // } - // } - // return true; - // } - if (is_node_fusable(node)) { - return true; - } - return false; - } - - //返回给定节点的子图 - Graph& get_subgraph(Node* node) { - AT_ASSERT(node->kind() == kind_); - return *node->g(attr::Subgraph); - } - - // 合并两个graph - void merge_fusion_groups(Node* consumer_group, Node* producer_group) { - // Now we have two fusion groups! - // Revert the fusion - place all inner nodes of producer back in the outer - // graph. - std::vector temporary_nodes; - Graph* producer_subgraph = &get_subgraph(producer_group); - - // Initialize a map of inner graph values to outer graph values - std::unordered_map inner_to_outer; - at::ArrayRef inner_inputs = producer_subgraph->inputs(); - at::ArrayRef outer_inputs = producer_group->inputs(); - for (size_t i = 0; i < inner_inputs.size(); ++i) { - inner_to_outer[inner_inputs[i]] = outer_inputs[i]; - } - - // Clone all nodes - for (auto inner : producer_subgraph->nodes()) { - Node* outer = block_->owningGraph()->createClone( - inner, [&](Value* k) -> Value* { return inner_to_outer.at(k); }); - for (size_t i = 0 ; i < outer->inputs().size(); i++){ - outer->input(i)->setType(inner->input(i)->type()); - } - outer->insertBefore(producer_group); - temporary_nodes.emplace_back(outer); - at::ArrayRef inner_outputs = inner->outputs(); - at::ArrayRef outer_outputs = outer->outputs(); - for (size_t i = 0; i < inner_outputs.size(); ++i) { - inner_to_outer[inner_outputs[i]] = outer_outputs[i]; - } - } - - // Replace uses of producer_group outputs and destroy the producer - at::ArrayRef subgraph_outputs = producer_subgraph->outputs(); - for (size_t i = 0; i < subgraph_outputs.size(); ++i) { - Value* outer_output = inner_to_outer.at(subgraph_outputs[i]); - producer_group->outputs()[i]->replaceAllUsesWith(outer_output); - update_global_context(producer_group->outputs()[i], outer_output); - } - - // Inline the temporary nodes into the first group - Graph* consumer_subgraph = &get_subgraph(consumer_group); - for (auto it = temporary_nodes.rbegin(); it != temporary_nodes.rend(); ++it) { - Node* node = *it; - Node* merged = merge_node_into_group(consumer_group, node); - // If any of the outputs are still used then we need to add them - at::ArrayRef outputs = node->outputs(); - for (size_t i = 0; i < outputs.size(); ++i) { - Value* output = outputs[i]; - if (output->uses().size() == 0) { - continue; - } - consumer_subgraph->registerOutput(merged->outputs()[i]); - Value* new_output = consumer_group->addOutput(); - output->replaceAllUsesWith(new_output); - update_global_context(output, new_output); - new_output->setType(output->type()); - } - node->destroy(); - } - update_global_list_size_map_node_key_context(producer_group, consumer_group); - producer_group->destroy(); - producer_group = nullptr; // Just to get a clear error in case someone uses it - } - - Node* merge_node_into_group(Node* group, Node* to_merge_node) { - AT_ASSERT(to_merge_node->kind() != kind_); - Graph& subgraph = get_subgraph(group); - - // map from nodes in the surrounding graph to parameters in the fusion - // group's subgraph that correspond to them - std::unordered_map inputs_map; - size_t i = 0; - size_t tensor_insert_idx = 0; - AT_ASSERT(group->inputs().size() == subgraph.inputs().size()); - for (auto input : group->inputs()) { - inputs_map[input] = subgraph.inputs()[i++]; - if (input->type()->isSubtypeOf(TensorType::get())) { - tensor_insert_idx = i; //真的要单独搞个tensor_index_idx 么? - } - } - - WithInsertPoint guard(*subgraph.nodes().begin()); - for (auto input : to_merge_node->inputs()) { - if (inputs_map.count(input) == 0) { - // TODO: we are following the convention for no good reason; - // we don't need tensor to come before any other inputs. - if (input->type()->isSubtypeOf(TensorType::get())) { - Value* in_group = subgraph.insertInput(tensor_insert_idx); - in_group->setType(input->type()); - inputs_map[input] = in_group; - group->insertInput(tensor_insert_idx, input); - tensor_insert_idx++; - } else if ( - // TODO: extend the supporting inputs here. - (input->type()->isSubtypeOf(FloatType::get()) && - input->node()->kind() != prim::Constant) || - (to_merge_node->kind() == aten::_grad_sum_to_size && - input->type()->isSubtypeOf(ListType::ofInts()))) { - Value* in_group = subgraph.addInput(); - in_group->setType(input->type()); - inputs_map[input] = in_group; - group->addInput(input); - } else if (input->node()->kind() == prim::Constant) { - // inline the constants directly in the body of the fused group. - Node* in_const = - subgraph.createClone(input->node(), [](Value*) -> Value* { - throw std::runtime_error("unexpected input"); - }); - subgraph.insertNode(in_const); - inputs_map[input] = in_const->output(); - } else { - Value* in_group = subgraph.addInput(); - in_group->setType(input->type()); - inputs_map[input] = in_group; - group->addInput(input); - } - } - } - - // for (auto input : to_merge_node->inputs()) { - // update_inputs_map(to_merge_node, input); - // } - - // copy n into the graph, remapping its inputs to internal nodes - Node* in_graph = subgraph.createClone( - to_merge_node, [&](Value* k) -> Value* { return inputs_map[k]; }, true); - - at::ArrayRef inputs = group->inputs(); - for (size_t i = 0; i < to_merge_node->outputs().size(); ++i) { - auto it = std::find(inputs.begin(), inputs.end(), to_merge_node->outputs()[i]); - if (it != inputs.end()) { - size_t p = it - inputs.begin(); - group->removeInput(p); - subgraph.inputs()[p]->replaceAllUsesWith(in_graph->outputs()[i]); - subgraph.eraseInput(p); - } - } - return subgraph.insertNode(in_graph); - } - - //将node转化为一个subgraph - Node* create_singleton_fusion_group(Node* n) { - Node* group = block_->owningGraph()->createWithSubgraph(kind_); - // propogate position information for the new node so we can always - // have a valid mapping - group->insertBefore(n); - Node* mergedNode = merge_node_into_group(group, n); - if (mergedNode->outputs().size() == 1) { - get_subgraph(group).registerOutput(mergedNode->output()); - Value* sel = group->addOutput(); - sel->copyMetadata(n->output()); - n->replaceAllUsesWith(group); - update_global_context(n->output(), sel); - //fix bug: handle situation when node has more than one output situation. - } else { - for (size_t index = 0; index < mergedNode->outputs().size(); index++) { - get_subgraph(group).registerOutput(mergedNode->outputs().at(index)); - Value* new_value = group->insertOutput(index)->copyMetadata(n->outputs().at(index)); - n->outputs().at(index)->replaceAllUsesWith(new_value); - update_global_context(n->outputs().at(index), new_value); - } - } - update_global_list_size_map_node_key_context(n, group); - n->destroy(); - return group; - } - - at::optional try_fuse(Node* consumer, Value* producer) { - - LOG(INFO) << "[try_fuse] consumer: " << node_info(consumer); - LOG(INFO) << "[try_fuse] producer: " << node_info(producer->node()); - - bool shouldFuse = - //TODO: check carefully later - is_node_fusable(consumer, producer->node()) && - // Rearrange nodes such that all uses of producer are after the - // consumer. Fusion will rewrite those later uses to use the version of - // producer generated by the fused blob. In this case, producer becomes - // an output of the fusion group. - aliasDb_->moveBeforeTopologicallyValid(producer->node(), consumer); - - if (producer->node()->kind() == prim::Constant) { - shouldFuse = true; - } - - if (!shouldFuse) { - LOG(INFO) << "[try_fuse Fail] should not fuse"; - return at::nullopt; - } - - Node* group = consumer; - if (producer->node()->kind() == kind_) { - if (consumer->kind() != kind_) { - group = create_singleton_fusion_group(consumer); - // should not update here cause consumer has destroyed. - // update_global_list_size_map_node_key_context(consumer, group); - } - merge_fusion_groups(group, producer->node()); - LOG(INFO) << "[try_fuse Success] FusionGroup is: " << node_info(group); - return group; - } - - // TODO: pay attention here. we should check multi output situation carefully. - if (producer->node()->outputs().size() != 1 && - !all_users_are_this_cunsumer(consumer, producer)) { - LOG(INFO) << "[try_fuse Fail] Should not fuse, producer output sizes: " << producer->node()->outputs().size() - << ", and is all_users_are_this_cunsumer: " << all_users_are_this_cunsumer(consumer, producer); - return at::nullopt; - } - - if (consumer->kind() != kind_) { - group = create_singleton_fusion_group(consumer); - // should not update here cause consumer has destroyed. - // update_global_list_size_map_node_key_context(consumer, group); - } - - Node* merged = merge_node_into_group(group, producer->node()); - //support for constant input. cause we copy the input. no need to replace this. - //TODO: pay attention here. constant handle should be careful. - if (producer->uses().size() != 0 && - producer->node()->kind() != prim::Constant) { - get_subgraph(group).registerOutput(merged->output()); - Value* new_producer = group->addOutput(); - new_producer->copyMetadata(producer); - producer->replaceAllUsesWith(new_producer); - update_global_context(producer, new_producer); - } - update_global_list_size_map_node_key_context(producer->node(), group); - if (producer->node()->kind() != prim::Constant) { - producer->node()->destroy(); - } - LOG(INFO) << "[try_fuse Success] FusionGroup is: " << node_info(group); - return group; - } - - value_list sort_reverse_topological(ArrayRef inputs) { - value_list result; - for (auto i : inputs) { - if ((i->node()->owningBlock() == block_) || - (i->node()->kind() == prim::Constant)) { - result.push_back(i); - } - } - // Sort in reverse topological order - std::sort(result.begin(), result.end(), [&](Value* a, Value* b) { - return a->node()->isAfter(b->node()); - }); - return result; - } - - // returns where to continue scanning, and whether any fusion was made - // todo 换条件 - std::pair scan_node(Node* consumer, const std::string list_construct) { - if (is_node_fusable(consumer)) { - value_list inputs = sort_reverse_topological(consumer->inputs()); - for (Value* producer : inputs) { - if ((list_construct == "input" && (producer->node()->kind() != prim::ListConstruct || consumer->kind() != prim::CudaFusionGroup)) || - (list_construct == "output" && (consumer->kind() != prim::ListUnpack || producer->node()->kind() != prim::CudaFusionGroup)) || - (producer->node()->kind() == prim::ListUnpack) || (consumer->kind() == prim::ListConstruct)) { - continue; - } - - at::optional fusion_group = try_fuse(consumer, producer); - if (fusion_group) { - // after fusion, consumer moves into a FusionGroup, so inputs is no - // longer valid so we rescan the new FusionGroup for more fusions... - return std::make_pair(fusion_group.value()->reverseIterator(), true); - } - } - } - return std::make_pair(++consumer->reverseIterator(), false); - } - - void refresh_aliasdb() { - aliasDb_ = torch::make_unique(graph_); - } - - void optimize_fused_graphs() { - for (Node* node : block_->nodes()) { - if (node->kind() != kind_) { - continue; - } - auto subgraph = node->g(attr::Subgraph); - EliminateDeadCode(subgraph); - EliminateCommonSubexpression(subgraph); - ConstantPooling(subgraph); - } - } - - void run(const std::string list_construct="") { - bool any_changed = true; - while (any_changed) { - any_changed = false; - refresh_aliasdb(); - for (auto it = block_->nodes().rbegin(); it != block_->nodes().rend();) { - bool changed = false; - std::tie(it, changed) = scan_node(*it, list_construct); - any_changed |= changed; - } - } - refresh_aliasdb(); - optimize_fused_graphs(); - - //TODO: should I add this??? - // for (Node* n : block_->nodes()) { - // removeOutputsUsedOnlyInSize(n); - // } - - for (Node* node : block_->nodes()) { - for (Block* sub_block : node->blocks()) { - PorosGraphSegment(sub_block, graph_, engine_).run(list_construct); - } - } - } -}; // struct PorosGraphSegment - -void gen_value_dyanamic_shape_of_tensorlist(torch::jit::Value* tensor_value, - size_t idx, - std::map> type_map) { - auto &_value_dynamic_shape_map = PorosGlobalContext::instance()._value_dynamic_shape_map; - - // 这里在的tensor_value地址可能和预热时的profile value地址一样,所以直接覆盖 - ValueDynamicShape dynamic_shape; - _value_dynamic_shape_map[tensor_value] = dynamic_shape; - _value_dynamic_shape_map[tensor_value].is_dynamic = false; - _value_dynamic_shape_map[tensor_value].max_shapes = type_map[idx][0]->sizes().concrete_sizes().value(); - _value_dynamic_shape_map[tensor_value].min_shapes = type_map[idx][0]->sizes().concrete_sizes().value(); - _value_dynamic_shape_map[tensor_value].opt_shapes = type_map[idx][0]->sizes().concrete_sizes().value(); - - // max - for (size_t i = 0; i < type_map[idx].size(); i++){ - std::vector tmp_max_shape = _value_dynamic_shape_map[tensor_value].max_shapes; - for(size_t j = 0; j < tmp_max_shape.size(); j++){ - _value_dynamic_shape_map[tensor_value].max_shapes[j] = std::max(tmp_max_shape[j], type_map[idx][i]->sizes()[j].value()); - } - } - - // min - for (size_t i = 0; i < type_map[idx].size(); i++){ - std::vector tmp_min_shape = _value_dynamic_shape_map[tensor_value].min_shapes; - for(size_t j = 0; j < tmp_min_shape.size(); j++){ - _value_dynamic_shape_map[tensor_value].min_shapes[j] = std::min(tmp_min_shape[j], type_map[idx][i]->sizes()[j].value()); - } - } - - ValueDynamicShape& shape = _value_dynamic_shape_map[tensor_value]; - for (size_t i = 0; i < shape.max_shapes.size(); ++i) { - if (shape.max_shapes[i] == shape.min_shapes[i] && shape.max_shapes[i] == shape.opt_shapes[i]) { - shape.sizes.push_back(shape.max_shapes[i]); - } else { - shape.sizes.push_back(-1); - shape.is_dynamic = true; - } - } -} - -// 此处为子图预判断 -// 作用:由于AdjustmentListTensorOutput、AdjustmentListTensorInput、AdjustmentScalarInput处理子图时会额外增加一些节点, -// 对于一些可预知的必然回退的子图(例如:unconst node不足、不支持的输出类型等),我们就不去处理这个子图,以减少不必要的节点数增加。 -bool cudafusion_should_be_handle(torch::jit::Node* node) { - // 如果传入是其他节点,直接返回false - if (node->kind() != torch::jit::prim::CudaFusionGroup) { - return false; - } - - int non_constant_node_num = 0; - std::shared_ptr subgraph = node->g(torch::jit::attr::Subgraph); - Block* subblock = subgraph->block(); - // 子图太小不处理 - int32_t unconst_threshold = PorosGlobalContext::instance().get_poros_options().unconst_ops_thres; - for (auto it = subblock->nodes().begin(); it != subblock->nodes().end(); ++it) { - if (it->kind() != torch::jit::prim::Constant) { - non_constant_node_num++; - if (non_constant_node_num > unconst_threshold) { - break; - } - } - } - if (non_constant_node_num <= unconst_threshold) { - LOG(WARNING) << "Subgraph: " << node_info_with_attr(node) << " size is too small, No tactics will be applied to it."; - return false; - } - return true; -} - -void AddPackAndUnpack(std::shared_ptr& group_graph, torch::jit::Value* value, size_t idx, torch::jit::Node* node, bool input=true){ - /* 在CudaFusionGroup内(ouput)/外(input),添加一个prim::ListUnpack;List[Tensor] -> Tensor、Tensor ... - 在CudaFusionGroup外(ouput)/内(input),添加一个prim::ListConstruct;Tensor、Tensor ... -> List[Tensor]*/ - - // 根据input or output 选择不同的map, 如果input中存在该list,则优先使用input;这样避免了append给output带来的引用问题。 - LIST_SIZE_MAP list_size_map = {}; //PorosGlobalContext::instance()._list_size_map._list_size_map_input; - TENSOR_LIST_TYPE_MAP list_tensor_type_map = {}; //PorosGlobalContext::instance()._list_size_map._list_tensor_type_map_input; - - if (input || PorosGlobalContext::instance()._list_size_map._list_size_map_input.count(value) != 0) { - list_size_map = PorosGlobalContext::instance()._list_size_map._list_size_map_input; - list_tensor_type_map = PorosGlobalContext::instance()._list_size_map._list_tensor_type_map_input; - if (!input) { - Node* input_node = list_size_map[value].begin()->first; - list_size_map[value][node] = list_size_map[value][input_node]; - list_tensor_type_map[value][node] = list_tensor_type_map[value][input_node]; - } - } - else { - list_size_map = PorosGlobalContext::instance()._list_size_map._list_size_map_output; - list_tensor_type_map = PorosGlobalContext::instance()._list_size_map._list_tensor_type_map_output; - } - - // 获取该tensorlist的长度 - int list_size = 0; - if (list_size_map.count(value) > 0) { - if (list_size_map[value].count(node) > 0) { - if (list_size_map[value][node].size() != 1) { - LOG(INFO) << "list " + value->debugName() << " has " << std::to_string(list_size_map[value].size()) << " lengths"; - return; - } - list_size = *list_size_map[value][node].begin(); - } - else { - LOG(INFO) << "node is not in list_size_map, value: %" << value->debugName() << ", node info:" << node_info(node); - throw c10::Error("node must be in list_size_map", ""); - } - } - else { - LOG(INFO) << "value is not in list_size_map, value: %" << value->debugName(); - throw c10::Error("value must be in list_size_map", ""); - } - if (list_size == 0) { - LOG(INFO) << "The length of the output list is 0: " << node_info(node); - return; - } - - // 新建一个unpack_node 和 pack_node - Node* unpack_node = group_graph->create(prim::ListUnpack, value); - Node* pack_node = group_graph->create(prim::ListConstruct, unpack_node->outputs()); - pack_node->output(0)->setType(value->type()); - std::vector guard_types; - - // 更新下,给后面的前置判断使用 - list_size_map[value][unpack_node] = {list_size}; - - if(input) { - pack_node->insertBefore(node); - unpack_node->insertBefore(pack_node); - } - else { - unpack_node->insertAfter(node); - pack_node->insertAfter(unpack_node); - } - - // 更新相关输入输出 - bool is_dynamic_shape = PorosGlobalContext::instance().get_poros_options().is_dynamic; - std::map> type_map = list_tensor_type_map[value][node]; - pack_node->replaceInput(0, unpack_node->output(0)); - unpack_node->replaceInput(0, value); - unpack_node->output(0)->setType(type_map[0][0]); - pack_node->input(0)->setType(type_map[0][0]); - guard_types.push_back(type_map[0][0]); - - if (!input) { - value->replaceAllUsesWith(pack_node->output(0)); - update_global_context(value, pack_node->output(0)); - unpack_node->replaceInput(0, value); - } - if (is_dynamic_shape && input) { - gen_value_dyanamic_shape_of_tensorlist(unpack_node->output(0), 0, type_map); - } - - for (int j = 0; j < list_size - 1; j++){ - unpack_node->insertOutput(j + 1); - pack_node->insertInput(j + 1, unpack_node->output(j + 1)); - unpack_node->output(j + 1)->setType(type_map[j + 1][0]); - pack_node->input(j + 1)->setType(type_map[j + 1][0]); - guard_types.push_back(type_map[j + 1][0]); - if (is_dynamic_shape && input) { - gen_value_dyanamic_shape_of_tensorlist(unpack_node->output(j + 1), j + 1, type_map); - } - } - - if (input) { - node->replaceInput(idx, pack_node->output(0)); - } - unpack_node->tys_(attr::types, guard_types); -} - -void AdjustmentListTensorInput(std::shared_ptr& group_graph, Block* block) { - /*把tensor list类型的输入纠正为多个tensor的输入,使其适配tensorrt的输入类型*/ - graph_node_list nodes = block->nodes(); - for(auto it = nodes.begin(); it != nodes.end(); it++){ - for (Block* subblock : it->blocks()) { - AdjustmentListTensorInput(group_graph, subblock); - } - if (it->kind() == prim::CudaFusionGroup) { - if (!cudafusion_should_be_handle(*it)) { - continue; - } - at::ArrayRef inputs = it->inputs(); - for (size_t i = 0; i < inputs.size(); i++){ - if(inputs[i]->type()->str() == "Tensor[]") { - LOG(INFO) << "Adjustment Tensor[] input %" << inputs[i]->debugName(); - AddPackAndUnpack(group_graph, inputs[i], i, *it, true); - } - } - } - } -} - -void AdjustmentListTensorOutput(std::shared_ptr& group_graph, Block* block) { - /*把tensor list类型的输出纠正为多个tensor的输入,使其适配tensorrt的输出类型*/ - graph_node_list nodes = block->nodes(); - for(auto it = nodes.begin(); it != nodes.end(); it++){ - for (Block* subblock : it->blocks()) { - AdjustmentListTensorOutput(group_graph, subblock); - } - if (it->kind() == prim::CudaFusionGroup) { - if (!cudafusion_should_be_handle(*it)) { - continue; - } - at::ArrayRef outputs = it->outputs(); - for (size_t i = 0; i < outputs.size(); i++){ - if(outputs[i]->type()->str() == "Tensor[]") { - LOG(INFO) << "Adjustment Tensor[] output %" << outputs[i]->debugName(); - AddPackAndUnpack(group_graph, outputs[i], i, *it, false); - } - } - } - } -} -// When cudafusiongroup subgraph input is int (or int[]) like: -// %1 : int = size(%x, %b) -// %4 : Tensor = prim::CudaFusionGroup(%1) -// or -// %1 : int[] = size(%x) -// %4 : Tensor = prim::CudaFusionGroup(%1) -// -// Then we insert aten::tensor and aten::IntImplicit (or prim::tolist) before the subgraph like: -// %1 : int = size(%x, %b) -// %2 : Tensor = aten::tensor(%1, %type, %device, %requires_grad) -// %3 : int = aten::IntImplicit(%2) -// %4 : Tensor = prim::CudaFusionGroup(%3) -// or -// %1 : int[] = size(%x) -// %2 : Tensor = aten::tensor(%1, %type, %device, %requires_grad) -// %3 : int = prim::tolist(%2, %dim, %type) -// %4 : Tensor = prim::CudaFusionGroup(%3) -// -// Finally, merge the aten::IntImplicit (or prim::tolist) into the cudafusiongroup subgraph. The int input has been replaced by tensor. -bool AddInputTensorandScalarimplict(std::shared_ptr& group_graph, torch::jit::Value* value, size_t idx, torch::jit::Node* node, IEngine* engine) { - bool value_type_is_list = (value->type()->kind() == c10::TypeKind::ListType); - int32_t list_size = 1; - LIST_SIZE_MAP list_size_map = {}; - if (value_type_is_list) { - // get list size - list_size_map = PorosGlobalContext::instance()._list_size_map._list_size_map_input; - if (list_size_map.count(value) > 0) { - if (list_size_map[value].count(node) > 0) { - if (list_size_map[value][node].size() != 1) { - LOG(WARNING) << "list " + value->debugName() << " has " << std::to_string(list_size_map[value].size()) << " lengths"; - return false; - } - list_size = *list_size_map[value][node].begin(); - } else { - LOG(WARNING) << "node is not in list_size_map, value: %" << value->debugName() << ", node info:" << node_info(node); - return false; - } - } else { - LOG(WARNING) << "value is not in list_size_map, value: %" << value->debugName(); - return false; - } - } - // 检查全局_int_intlist_values_map中有无当前scalar值 - std::map& int_intlist_values_map = PorosGlobalContext::instance()._int_intlist_values_map; - if (value->type()->isSubtypeOf(c10::ListType::ofInts()) || value->type()->kind() == c10::TypeKind::IntType) { - if (int_intlist_values_map.count(value) == 0) { - LOG(WARNING) << "can't find max min opt of int(or int[]) %" << value->debugName(); - return false; - } - } - - std::map& value_dynamic_shape_map = PorosGlobalContext::instance()._value_dynamic_shape_map; - auto fuser = PorosGraphSegment(group_graph->block(), group_graph, engine); - // 创建aten::tensor - torch::jit::Node* tensor_node = group_graph->create(torch::jit::aten::tensor); - tensor_node->insertBefore(node); - tensor_node->addInput(value); - // note: 没有setInsertPoint insertconstant默认到图的末尾插入节点 - // 但此处最好不要用setInsertPoint,当图发生变化导致point的点变化时候会出core - // 建议使用”insertConstant之后moveBerfore“来代替”setInsertPoint后insertConstant“的操作 - // group_graph->setInsertPoint(tensor_node); - // 创建aten::tensor dtype、device和requires_grad constant输入 - torch::jit::Value* type_value = nullptr; - c10::optional output_scalar_type; - if (value_type_is_list) { - if (value->type()->isSubtypeOf(c10::ListType::ofInts())) { - type_value = group_graph->insertConstant(c10::ScalarType::Long); - output_scalar_type = at::kLong; - } else { - type_value = group_graph->insertConstant(c10::ScalarType::Float); - output_scalar_type = at::kFloat; - } - } else { - if (value->type()->kind() == c10::TypeKind::IntType) { - type_value = group_graph->insertConstant(c10::ScalarType::Int); - output_scalar_type = at::kInt; - } else { - type_value = group_graph->insertConstant(c10::ScalarType::Float); - output_scalar_type = at::kFloat; - } - } - torch::jit::Value* device_value = nullptr; - c10::optional output_device; - if (PorosGlobalContext::instance().get_poros_options().device == Device::CUDA) { - device_value = group_graph->insertConstant(torch::Device(torch::DeviceType::CUDA, 0)); - output_device = torch::Device(at::kCUDA, 0); - } else { - torch::jit::IValue none_ivalue; - device_value = group_graph->insertConstant(none_ivalue); - output_device = torch::Device(at::kCPU); - } - torch::jit::Value* false_value = group_graph->insertConstant(false); - // 没有setinsertpoint,insertconstant默认到了图的末尾,需要将constant移到tensor_node之前 - type_value->node()->moveBefore(tensor_node); - device_value->node()->moveBefore(tensor_node); - false_value->node()->moveBefore(tensor_node); - - tensor_node->addInput(type_value); - tensor_node->addInput(device_value); - tensor_node->addInput(false_value); - // must set output type - TypePtr output_type = c10::TensorType::create(output_scalar_type, - output_device, - c10::SymbolicShape(std::vector>({list_size})), - std::vector({c10::Stride{0, true, 1}}), - false); - tensor_node->output(0)->setType(output_type); - // 更新value_dynamic_shape_map中aten::tensor output的max min opt值为int_intlist_values_map中的value对应的值。 - // 因为tensor_node->output(0)即将变为子图输入 - value_dynamic_shape_map[tensor_node->output(0)] = int_intlist_values_map[value]; - - // 创建scalar implicit node - // 如果是scalarlist - if (value_type_is_list) { - // 更新list_size_map中value在aten::tensor的list_size信息 - list_size_map[value][tensor_node] = {(int32_t)list_size}; - // int list create prim::tolist node - torch::jit::Node* tolist_node = group_graph->create(torch::jit::prim::tolist); - tolist_node->insertBefore(node); - torch::jit::Value* dim_val = group_graph->insertConstant(int(1)); - torch::jit::Value* type_val = nullptr; - if (value->type()->isSubtypeOf(c10::ListType::ofInts())) { - // int list - type_val = group_graph->insertConstant(int(0)); - } else { - // float list - type_val = group_graph->insertConstant(int(1)); - } - tolist_node->addInput(tensor_node->output(0)); - - dim_val->node()->moveBefore(tolist_node); - type_val->node()->moveBefore(tolist_node); - - tolist_node->addInput(dim_val); - tolist_node->addInput(type_val); - - if (value->type()->isSubtypeOf(c10::ListType::ofInts())) { - tolist_node->output(0)->setType(c10::ListType::ofInts()); - } else { - tolist_node->output(0)->setType(c10::ListType::ofFloats()); - } - node->replaceInput(idx, tolist_node->output(0)); - - // 手动更新map - list_size_map[tolist_node->output(0)][tolist_node] = {(int32_t)list_size}; - list_size_map[tolist_node->output(0)][node] = {(int32_t)list_size}; - int_intlist_values_map[tolist_node->output(0)] = int_intlist_values_map[value]; - - // 把tolist merge进子图中 - fuser.refresh_aliasdb(); - fuser.merge_node_into_group(node, type_val->node()); - fuser.merge_node_into_group(node, dim_val->node()); - fuser.merge_node_into_group(node, tolist_node); - fuser.refresh_aliasdb(); - fuser.optimize_fused_graphs(); - - // 如果输入是scalar - } else { - // int创建intimplicit - torch::jit::Node* scalar_implicit_node = nullptr; - if (value->type()->kind() == c10::TypeKind::IntType) { - torch::jit::Node* intimplicit_node = group_graph->create(torch::jit::aten::IntImplicit, tensor_node->output(0)); - intimplicit_node->output(0)->setType(c10::IntType::get()); - intimplicit_node->insertBefore(node); - node->replaceInput(idx, intimplicit_node->output(0)); - scalar_implicit_node = intimplicit_node; - } else { - // float创建FloatImplicit - torch::jit::Node* floatimplicit_node = group_graph->create(torch::jit::aten::FloatImplicit, tensor_node->output(0)); - floatimplicit_node->output(0)->setType(c10::FloatType::get()); - floatimplicit_node->insertBefore(node); - node->replaceInput(idx, floatimplicit_node->output(0)); - scalar_implicit_node = floatimplicit_node; - } - // 更新int_intlist_values_map - int_intlist_values_map[scalar_implicit_node->output(0)] = int_intlist_values_map[value]; - fuser.refresh_aliasdb(); - fuser.try_fuse(node, node->input(idx)); - fuser.refresh_aliasdb(); - fuser.optimize_fused_graphs(); - } - return true; -} - -// 当子图输出是scalar(或scalar list)类型时, -// 创建aten::tensor与aten::IntImplicit(或prim::tolist) -// 然后将aten::tensor融合到子图中去 -bool AddOutputTensorandScalarimplict(std::shared_ptr& group_graph, torch::jit::Value* value, size_t idx, torch::jit::Node*& node, IEngine* engine) { - bool value_type_is_list = (value->type()->kind() == c10::TypeKind::ListType); - size_t list_size = 1; - LIST_SIZE_MAP list_size_map = {}; - if (value_type_is_list) { - // get list size - if (PorosGlobalContext::instance()._list_size_map._list_size_map_input.count(value) != 0) { - list_size_map = PorosGlobalContext::instance()._list_size_map._list_size_map_input; - Node* input_node = list_size_map[value].begin()->first; - list_size_map[value][node] = list_size_map[value][input_node]; - } - else { - list_size_map = PorosGlobalContext::instance()._list_size_map._list_size_map_output; - } - if (list_size_map.count(value) > 0) { - if (list_size_map[value].count(node) > 0) { - if (list_size_map[value][node].size() != 1) { - LOG(WARNING) << "list " + value->debugName() << " has " << std::to_string(list_size_map[value].size()) << " lengths"; - return false; - } - list_size = *list_size_map[value][node].begin(); - } else { - LOG(WARNING) << "node is not in list_size_map, value: %" << value->debugName() << ", node info:" << node_info(node); - return false; - } - } else { - LOG(WARNING) << "value is not in list_size_map, value: %" << value->debugName(); - return false; - } - } - // 检查全局_int_intlist_values_map中有无当前scalar值 - std::map& int_intlist_values_map = PorosGlobalContext::instance()._int_intlist_values_map; - if (value->type()->isSubtypeOf(c10::ListType::ofInts()) || value->type()->kind() == c10::TypeKind::IntType) { - if (int_intlist_values_map.count(value) == 0) { - LOG(WARNING) << "can't find max min opt of int(or int[]) %" << value->debugName(); - return false; - } - } - - std::map& value_dynamic_shape_map = PorosGlobalContext::instance()._value_dynamic_shape_map; - auto fuser = PorosGraphSegment(group_graph->block(), group_graph, engine); - // 创建aten::tensor - torch::jit::Node* tensor_node = group_graph->create(torch::jit::aten::tensor); - tensor_node->insertAfter(node); - tensor_node->addInput(value); - // 创建aten::tensor dtype、device和requires_grad constant输入 - torch::jit::Value* type_value = nullptr; - c10::optional output_scalar_type; - if (value_type_is_list) { - if (value->type()->isSubtypeOf(c10::ListType::ofInts())) { - type_value = group_graph->insertConstant(c10::ScalarType::Long); - output_scalar_type = at::kLong; - } else { - type_value = group_graph->insertConstant(c10::ScalarType::Float); - output_scalar_type = at::kFloat; - } - } else { - if (value->type()->kind() == c10::TypeKind::IntType) { - type_value = group_graph->insertConstant(c10::ScalarType::Int); - output_scalar_type = at::kInt; - } else { - type_value = group_graph->insertConstant(c10::ScalarType::Float); - output_scalar_type = at::kFloat; - } - } - torch::jit::Value* device_value = nullptr; - c10::optional output_device; - if (PorosGlobalContext::instance().get_poros_options().device == Device::CUDA) { - device_value = group_graph->insertConstant(torch::Device(torch::DeviceType::CUDA, 0)); - output_device = torch::Device(at::kCUDA, 0); - } else { - torch::jit::IValue none_ivalue; - device_value = group_graph->insertConstant(none_ivalue); - output_device = torch::Device(at::kCPU); - } - torch::jit::Value* false_value = group_graph->insertConstant(false); - - type_value->node()->moveBefore(tensor_node); - device_value->node()->moveBefore(tensor_node); - false_value->node()->moveBefore(tensor_node); - - tensor_node->addInput(type_value); - tensor_node->addInput(device_value); - tensor_node->addInput(false_value); - - // must set output type - TypePtr output_type = c10::TensorType::create(output_scalar_type, - output_device, - c10::SymbolicShape(std::vector>({list_size})), - std::vector({c10::Stride{0, true, 1}}), - false); - tensor_node->output(0)->setType(output_type); - - value_dynamic_shape_map[tensor_node->output(0)] = int_intlist_values_map[value]; - - // 创建scalar implicit node - // 如果输入是scalarlist - torch::jit::Node* tolist_node = nullptr; - torch::jit::Node* scalar_implicit_node = nullptr; - if (value_type_is_list) { - // 更新list_size_map中value在aten::tensor子图的list_size信息 - list_size_map[value][tensor_node] = {(int32_t)list_size}; - // int list create prim::tolist node - tolist_node = group_graph->create(torch::jit::prim::tolist); - tolist_node->insertAfter(tensor_node); - tolist_node->addInput(tensor_node->output(0)); - torch::jit::Value* dim_val = group_graph->insertConstant(int(1)); - torch::jit::Value* type_val = nullptr; - if (value->type()->isSubtypeOf(c10::ListType::ofInts())) { - // int list - type_val = group_graph->insertConstant(int(0)); - } else { - // float list - type_val = group_graph->insertConstant(int(1)); - } - - dim_val->node()->moveBefore(tolist_node); - type_val->node()->moveBefore(tolist_node); - - tolist_node->addInput(dim_val); - tolist_node->addInput(type_val); - - if (value->type()->isSubtypeOf(c10::ListType::ofInts())) { - tolist_node->output(0)->setType(c10::ListType::ofInts()); - } else { - tolist_node->output(0)->setType(c10::ListType::ofFloats()); - } - value->replaceAllUsesAfterNodeWith(tolist_node, tolist_node->output(0)); - - list_size_map[tolist_node->output(0)][tolist_node] = {(int32_t)list_size}; - // list_size_map中value有node概念,需要一个一个更新 - torch::jit::use_list tolist_node_user = tolist_node->output(0)->uses(); - for (size_t u = 0; u < tolist_node_user.size(); u++) { - list_size_map[tolist_node->output(0)][tolist_node_user[u].user] = {(int32_t)list_size}; - } - int_intlist_values_map[tolist_node->output(0)] = int_intlist_values_map[value]; - } else { - // int create intimplicit node - if (value->type()->kind() == c10::TypeKind::IntType) { - torch::jit::Node* intimplicit_node = group_graph->create(torch::jit::aten::IntImplicit, tensor_node->output(0)); - intimplicit_node->output(0)->setType(c10::IntType::get()); - intimplicit_node->insertAfter(tensor_node); - scalar_implicit_node = intimplicit_node; - } else { - // float create FloatImplicit node - torch::jit::Node* floatimplicit_node = group_graph->create(torch::jit::aten::FloatImplicit, tensor_node->output(0)); - floatimplicit_node->output(0)->setType(c10::FloatType::get()); - floatimplicit_node->insertAfter(tensor_node); - scalar_implicit_node = floatimplicit_node; - } - value->replaceAllUsesAfterNodeWith(scalar_implicit_node, scalar_implicit_node->output(0)); - // 更新int_intlist_values_map - int_intlist_values_map[scalar_implicit_node->output(0)] = int_intlist_values_map[value]; - } - // 为aten::tensor 创造子图,更新全局map,最后与node fuser - fuser.refresh_aliasdb(); - torch::jit::Node* subgraph_node = fuser.create_singleton_fusion_group(tensor_node); - fuser.merge_node_into_group(subgraph_node, type_value->node()); - fuser.merge_node_into_group(subgraph_node, device_value->node()); - fuser.merge_node_into_group(subgraph_node, false_value->node()); - // list_size_map只要更换节点就需要更新 - if (value_type_is_list) { - list_size_map[value][subgraph_node] = {(int32_t)list_size}; - } - value_dynamic_shape_map[subgraph_node->output(0)] = int_intlist_values_map[value]; - fuser.try_fuse(subgraph_node, subgraph_node->input(0)); - // 由于用了aten::tensor构造的子图来fuse,之前node的子图已经消失,需更新node为新融合的子图 - node = subgraph_node; - fuser.refresh_aliasdb(); - fuser.optimize_fused_graphs(); - return true; -} - -// 将子图的scalar输入转成tensor输入 -bool adjust_scalar_input(std::shared_ptr& group_graph, Block* block, IEngine* engine) { - bool changed = false; - graph_node_list nodes = block->nodes(); - for(auto it = nodes.begin(); it != nodes.end(); ) { - Node* current_node = *it; - it++; - for (Block* subblock : current_node->blocks()) { - changed |= adjust_scalar_input(group_graph, subblock, engine); - } - if (current_node->kind() == prim::CudaFusionGroup) { - if (!cudafusion_should_be_handle(current_node)) { - continue; - } - at::ArrayRef subgraph_node_inputs = current_node->inputs(); - for (size_t i = 0; i < subgraph_node_inputs.size(); i++) { - // todo: support float and float[] - // mark by tsq 0713: loop中的scalar input可能会有问题,因为其记录的max min opt不一定真实,但目前没有遇到此类问题。 - if(subgraph_node_inputs[i]->type()->str() == "int" || subgraph_node_inputs[i]->type()->str() == "int[]") { - LOG(INFO) << "Adjustment subgraph: " << node_info_with_attr(current_node) << " scalar input %" << subgraph_node_inputs[i]->debugName(); - std::string origin_input_debugname = subgraph_node_inputs[i]->debugName(); - if (AddInputTensorandScalarimplict(group_graph, subgraph_node_inputs[i], i, current_node, engine)) { - LOG(INFO) << "Adjustment scalar input %" << origin_input_debugname << " to tensor %" - << subgraph_node_inputs[i]->debugName() << " succeed!"; - changed = true; - } else { - LOG(WARNING) << "Adjustment scalar input %" << origin_input_debugname << " failed!"; - } - } - } - } - } - return changed; -} - -void AdjustmentScalarInput(std::shared_ptr& group_graph, Block* block, IEngine* engine) { - bool changed = false; - changed = adjust_scalar_input(group_graph, block, engine); - if (changed) { - EliminateDeadCode(group_graph); - EliminateCommonSubexpression(group_graph); - ConstantPooling(group_graph); - } -} - -// 将子图的scalar输出转为tensor输出 -bool adjust_scalar_output(std::shared_ptr& group_graph, Block* block, IEngine* engine) { - /*把tensor list类型的输入纠正为多个tensor的输入,使其适配tensorrt的输入类型*/ - bool changed = false; - graph_node_list nodes = block->nodes(); - for(auto it = nodes.begin(); it != nodes.end(); ) { - Node* current_node = *it; - it++; - for (Block* subblock : current_node->blocks()) { - changed |= adjust_scalar_output(group_graph, subblock, engine); - } - if (current_node->kind() == prim::CudaFusionGroup) { - if (!cudafusion_should_be_handle(current_node)) { - continue; - } - - for (size_t i = 0; i < current_node->outputs().size(); i++) { - if (current_node->output(i)->type()->str() == "int" || current_node->output(i)->type()->str() == "int[]") { - // todo: support float and float[] - LOG(INFO) << "Adjustment subgraph: " << node_info_with_attr(current_node) << " scalar output %" << current_node->output(i)->debugName(); - std::string origin_output_debugname = current_node->output(i)->debugName(); - if (AddOutputTensorandScalarimplict(group_graph, current_node->output(i), i, current_node, engine)) { - LOG(INFO) << "Adjustment scalar output %" << origin_output_debugname << " to tensor %" - << current_node->output(i)->debugName() << " succeed!"; - changed = true; - // 更新scalar output后子图会更新,在新的子图上继续寻找scalar output,直到所有输出都不是scalar。 - i = 0; - } else { - LOG(WARNING) << "Adjustment scalar output %" << origin_output_debugname << " failed!"; - } - } - } - } - } - return changed; -} - -void AdjustmentScalarOutput(std::shared_ptr& group_graph, Block* block, IEngine* engine) { - bool changed = false; - changed = adjust_scalar_output(group_graph, block, engine); - if (changed) { - EliminateDeadCode(group_graph); - EliminateCommonSubexpression(group_graph); - ConstantPooling(group_graph); - } -} - -void peephole_optimize_shape_expressions(Block* block) { - graph_node_list nodes = block->nodes(); - for (auto it = nodes.begin(); it != nodes.end(); ++it) { - Node* node = *it; - for (Block* subblock : node->blocks()) { - peephole_optimize_shape_expressions(subblock); - } - if (node->kind() == prim::BroadcastSizes) { - // Remove no-op broadcasts. - if (node->inputs().size() == 1) { - node->output()->replaceAllUsesWith(node->input()); - it.destroyCurrent(); - continue; - } - // Deduplicate inputs, but use their unique() values to ensure - // this process only depends on the graph. - std::map unique_to_value; - for (Value* input : node->inputs()) { - unique_to_value.emplace(input->unique(), input); - } - if (unique_to_value.size() != node->inputs().size()) { - std::vector inputs; - inputs.reserve(unique_to_value.size()); - for (auto& entry : unique_to_value) { - inputs.push_back(entry.second); - } - if (inputs.size() == 1) { - node->output()->replaceAllUsesWith(inputs[0]); - } else { - WithInsertPoint insert_guard{node}; - node->output()->replaceAllUsesWith(broadcast_sizes(inputs)); - } - it.destroyCurrent(); - --it; // Revisit the node with deduplicated inputs - continue; - } - // Remove compose simple chains of broadcasts into a single node. - const auto& uses = node->output()->uses(); - if (uses.size() == 1 && uses[0].user->kind() == prim::BroadcastSizes) { - Node* user = uses[0].user; - user->removeInput(uses[0].offset); - // NB: we don't care about deduplication in here, as we will visit user - // later. - for (Value* i : node->inputs()) { - user->addInput(i); - } - it.destroyCurrent(); - } - } - } -} // peephole_optimize_shape_expressions - -void guard_fusion_group(Node* fusion) { - // Fixup types of the subgraph inputs - std::vector guard_types; - std::vector inputs_to_check; - for (Value* input : fusion->inputs()) { - // We only check inputs of the fusion group and expect NNC to infer - // intermediates and outputs shapes - if (!input->type()->cast()) { - continue; - } - - // note: modified from original implementation, we are guarding fusion - // outputs - if (input->node()->kind() == prim::Constant) { - continue; - } - inputs_to_check.push_back(input); - guard_types.push_back(input->type()); - } - if (!inputs_to_check.size()) { - return; - } - - Node* typecheck_node = fusion->owningGraph() - //this is not right, i should register my own type to torchscrilpt - //->create(prim::CudaFusionGuard, inputs_to_check, 1) - ->create(prim::FusionGroup, inputs_to_check, 1) - ->insertBefore(fusion); - // fix output to BoolType - typecheck_node->output()->setType(BoolType::get()); - Value* typecheck_result = typecheck_node->output(); - typecheck_node->tys_(attr::types, guard_types); - - std::unordered_map typechecked_inputs; - - // Insert if block - Node* versioning_if = - fusion->owningGraph() - ->create(prim::If, {typecheck_result}, fusion->outputs().size()) - ->insertAfter(typecheck_node); - for (size_t idx = 0; idx < fusion->outputs().size(); ++idx) { - versioning_if->output(idx)->setType(fusion->output(idx)->type()); - fusion->output(idx)->replaceAllUsesWith(versioning_if->output(idx)); - } - Block* true_block = versioning_if->addBlock(); - Block* false_block = versioning_if->addBlock(); - - // Fill in the false block. It should contain the unoptimized - // copy of the fused subgraph. - auto& subgraph = *fusion->g(attr::Subgraph); - WithInsertPoint guard(false_block->return_node()); - const std::vector subgraph_outputs = - insertGraph(*fusion->owningGraph(), subgraph, fusion->inputs()); - for (Value* output : subgraph_outputs) { - false_block->registerOutput(output); - } - - // types get copied to the fallback graph, so remove specializations before - // replacing - // TODO: this is not exposed here, I need to remove that before inserting the - // graph - // removeTensorTypeSpecializations(false_block); - replaceBlockWithFallbackGraph(false_block, fusion->inputs()); - - // Fill in the true block. It has all inputs type-checked and its - // body should be the fusion group node. - fusion->moveBefore(true_block->return_node()); - for (Value* output : fusion->outputs()) { - true_block->registerOutput(output); - } -} // guard_fusion_group - -void guard_fusion_groups(Block* block) { - std::vector fusions; - for (Node* n : block->nodes()) { - for (Block* b : n->blocks()) { - guard_fusion_groups(b); - } - if (n->kind() == prim::CudaFusionGroup) { - fusions.push_back(n); - } - } - for (Node* fusion : fusions) { - guard_fusion_group(fusion); - } -} // guard_fusion_groups - -} // anonymous namespace - -void graph_segment(std::shared_ptr& graph, IEngine* engine) { - - GRAPH_DUMP("before PorosGraphSegment Graph: ", graph); - PorosGraphSegment(graph->block(), graph, engine).run(); - GRAPH_DUMP("after PorosGraphSegment Graph: ", graph); - //guard_fusion_groups(graph->block()); - - //necessary passes after segmentation - { - torch::jit::EliminateCommonSubexpression(graph); - torch::jit::EliminateDeadCode(graph); - peephole_optimize_shape_expressions(graph->block()); - torch::jit::RemoveTensorTypeSpecializations(graph); - GRAPH_DUMP("after necessary pass Graph: ", graph); - } - - //necessary adjustmentations after segmentation - { - AdjustmentListTensorInput(graph, graph->block()); - PorosGraphSegment(graph->block(), graph, engine).run("input"); - GRAPH_DUMP("after AdjustmentListTensorInput Graph: ", graph); - AdjustmentListTensorOutput(graph, graph->block()); - PorosGraphSegment(graph->block(), graph, engine).run("output"); - GRAPH_DUMP("after AdjustmentListTensorOutput Graph: ", graph); - AdjustmentScalarInput(graph, graph->block(), engine); - GRAPH_DUMP("after AdjustmentScalarInput Graph: ", graph); - AdjustmentScalarOutput(graph, graph->block(), engine); - GRAPH_DUMP("after AdjustmentScalarOutput Graph: ", graph); - } -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/compile/graph_segment.h b/poros/poros/compile/graph_segment.h deleted file mode 100644 index 2498edada83..00000000000 --- a/poros/poros/compile/graph_segment.h +++ /dev/null @@ -1,43 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file graph_segment.h -* @author tianjinjin@baidu.com -* @date Thu Mar 18 14:33:54 CST 2021 -* @brief -**/ - -#pragma once - -//pytorch -#include - -#include "poros/engine/iengine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/** - * @brief graph_segment fullfil the segmentation func of given graph - * @param [in/out] graph : the graph to be segmented - * @param [in] engine : backend engine - * @return - **/ -void graph_segment(std::shared_ptr& graph, IEngine* engine); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/compile/ivalues_analysis.cpp b/poros/poros/compile/ivalues_analysis.cpp deleted file mode 100644 index 55c261663fd..00000000000 --- a/poros/poros/compile/ivalues_analysis.cpp +++ /dev/null @@ -1,930 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file ivalues_analysis.cpp -* @author tianjinjin@baidu.com -* @date Fri Apr 23 11:41:59 CST 2021 -* @brief -**/ -#include "poros/compile/ivalues_analysis.h" - -#include - -#include -#include -#include -#include //check -#include //to get executioncontext -#include //for get numprofileruns - -#include "poros/util/poros_util.h" -#include "poros/context/poros_global.h" - -//I have to copy the ProfileOp function here -namespace torch { -namespace jit { - -const Symbol ProfileOp::Kind = ::c10::prim::profile; -void ProfileOp::cloneFrom(torch::jit::Node* other_) { - torch::jit::Node::cloneFrom(other_); - auto other = other_->cast(); - this->callback_ = other->getCallback(); -} - -torch::jit::Node* ProfileOp::allocNewInstance(torch::jit::Graph* g) { - return new ProfileOp(g, {nullptr}); -} - -} // namespace jit -} // namespace torch - -namespace baidu { -namespace mirana { -namespace poros { - -IvalueAnalysis::IvalueAnalysis(std::shared_ptr g) - : profiled_graph_(std::move(g)), profiling_count_(torch::jit::getNumProfiledRuns()) {} - -void IvalueAnalysis::insert_input_listsize_profile(torch::jit::Node* node, size_t offset) { - torch::jit::Value* input_value = node->input(offset); - - // 创建profile node - torch::jit::ProfileOp *pn = create_profile_node(nullptr, {input_value}); - auto pno = pn->addOutput(); - pn->ty_(torch::jit::attr::profiled_type, input_value->type()); - pno->setType(input_value->type()); - - std::function list_size_profile = [this, pno, node](torch::jit::Stack& stack){ - int64_t frame_id = 0; - torch::jit::pop(stack, frame_id); - c10::IValue ivalue; - torch::jit::pop(stack, ivalue); - std::lock_guard lock(this->mutex_); - if (ivalue.isList()) { - auto input_value = pno->node()->input(0); - //not exist yet, insert it - if (_list_size_map._list_size_map_input.count(input_value) == 0) { - std::set int_set{(int32_t)ivalue.toListRef().size()}; - _list_size_map._list_size_map_input[input_value][node] = int_set; - if (ivalue.isTensorList()) { - auto tl = ivalue.toTensorList(); - std::map> type_map; - for(size_t i = 0; i < ivalue.toListRef().size(); i++){ - auto tlty = torch::jit::tensorTypeInCurrentExecutionContext(tl[i]); - type_map[i] = {tlty}; - } - _list_size_map._list_tensor_type_map_input[input_value][node] = type_map; - } - } - else { - if (_list_size_map._list_size_map_input[input_value].count(node) == 0) { - std::set int_set{(int32_t)ivalue.toListRef().size()}; - _list_size_map._list_size_map_input[input_value][node] = int_set; - } - else { - _list_size_map._list_size_map_input[input_value][node].insert(ivalue.toListRef().size()); - } - if (ivalue.isTensorList()) { - auto tl = ivalue.toTensorList(); - std::map> &type_map = _list_size_map._list_tensor_type_map_input[input_value][node]; - for(size_t i = 0; i < ivalue.toListRef().size(); i++) { - auto tlty = torch::jit::tensorTypeInCurrentExecutionContext(tl[i]); - type_map[i].push_back(tlty); - } - } - } - // extract int[] values to map - if (input_value->type()->isSubtypeOf(c10::ListType::ofInts()) && - input_value->node()->kind() != torch::jit::prim::Constant && ivalue.isIntList()) { - auto& value_vec_map = _int_intlist_values_per_frame[frame_id]; - // extract int[] - std::vector int_vec; - c10::List c10_int_list = ivalue.toIntList(); - for (int64_t i : c10_int_list) { - int_vec.push_back(i); - } - //not exist yet, insert it - if (value_vec_map.count(input_value) == 0) { - std::vector> int_vec_vec; - int_vec_vec.push_back(int_vec); - value_vec_map.insert({input_value, int_vec_vec}); - } else { - value_vec_map[input_value].push_back(int_vec); - } - } - } - torch::jit::push(stack, ivalue); - }; - pn->setCallback(list_size_profile); - pn->insertBefore(node); - node->replaceInput(offset, pn->output()); -} - -void IvalueAnalysis::insert_number_eval_profile(torch::jit::Node* node, size_t offset) { - torch::jit::Value* input_value = node->input(offset); - // 创建profile node - torch::jit::ProfileOp *pn = create_profile_node(nullptr, {input_value}); - auto pno = pn->addOutput(); - pn->ty_(torch::jit::attr::profiled_type, input_value->type()); - pno->setType(input_value->type()); - - std::function int_intlist_profile = [this, input_value](torch::jit::Stack& stack) { - int64_t frame_id = 0; - torch::jit::pop(stack, frame_id); - c10::IValue ivalue; - torch::jit::pop(stack, ivalue); - std::lock_guard lock(this->mutex_); - if (ivalue.isInt()) { - auto& value_vec_map = _int_intlist_values_per_frame[frame_id]; - // extract int - std::vector int_vec; - int_vec.push_back(ivalue.toInt()); - //not exist yet, insert it - if (value_vec_map.count(input_value) == 0) { - std::vector> int_vec_vec; - int_vec_vec.push_back(int_vec); - value_vec_map.insert({input_value, int_vec_vec}); - } else { - value_vec_map[input_value].push_back(int_vec); - } - } - // passing t through - torch::jit::push(stack, ivalue); - }; - pn->setCallback(int_intlist_profile); - pn->insertBefore(node); - node->replaceInput(offset, pn->output()); -} - -void IvalueAnalysis::insert_output_listsize_profile(torch::jit::Node* node, size_t offset) { - torch::jit::Value* output_value = node->output(offset); - //watch this value - auto eval_pn = create_profile_node(nullptr, {output_value}); - auto pno = eval_pn->addOutput(); - eval_pn->ty_(torch::jit::attr::profiled_type, output_value->type()); - pno->setType(output_value->type()); - - //do we need outout? and change the input of prim::If? - std::function eval_profiler = [this, pno, node](torch::jit::Stack& stack) { - int64_t frame_id = 0; - torch::jit::pop(stack, frame_id); - c10::IValue ivalue; - torch::jit::pop(stack, ivalue); - std::lock_guard lock(this->mutex_); - - if (ivalue.isList()) { - //not exist yet, insert it - auto input_value = pno->node()->input(0); - if (_list_size_map._list_size_map_output.count(input_value) == 0) { - std::set int_set{(int32_t)ivalue.toListRef().size()}; - _list_size_map._list_size_map_output[input_value][node] = int_set; - - if (ivalue.isTensorList()) { - auto tl = ivalue.toTensorList(); - std::map> type_map; - for(size_t i = 0; i < ivalue.toListRef().size(); i++){ - auto tlty = torch::jit::tensorTypeInCurrentExecutionContext(tl[i]); - type_map[i] = {tlty}; - } - _list_size_map._list_tensor_type_map_output[input_value][node] = type_map; - } - } - else { - if (_list_size_map._list_size_map_output[input_value].count(node) == 0) { - std::set int_set{(int32_t)ivalue.toListRef().size()}; - _list_size_map._list_size_map_output[input_value][node] = int_set; - } - else { - _list_size_map._list_size_map_output[input_value][node].insert(ivalue.toListRef().size()); - } - - if (ivalue.isTensorList()) { - auto tl = ivalue.toTensorList(); - std::map> &type_map = _list_size_map._list_tensor_type_map_output[input_value][node]; - for(size_t i = 0; i < ivalue.toListRef().size(); i++) { - auto tlty = torch::jit::tensorTypeInCurrentExecutionContext(tl[i]); - type_map[i].push_back(tlty); - } - } - } - } - torch::jit::push(stack, ivalue); - }; - eval_pn->setCallback(eval_profiler); - eval_pn->insertAfter(node); -} - -//TODO: check out the difference between the ProfileIValueOp and ProfileOp -torch::jit::ProfileOp* IvalueAnalysis::create_profile_node( - const std::function& fp, - at::ArrayRef inputs) { - auto pn = new torch::jit::ProfileOp(profiled_graph_.get(), fp); - for (auto in : inputs) { - pn->addInput(in); - } - return pn; -} - -/* -torch::jit::ProfileOptionalOp* IvalueAnalysis::create_profile_optional_node( - const std::function& fp, - at::ArrayRef inputs) { - auto pn = new torch::jit::ProfileOptionalOp(profiled_graph_.get(), fp); - pn->i_(torch::jit::attr::num_present, 0); - pn->i_(torch::jit::attr::num_none, 0); - for (auto in : inputs) { - pn->addInput(in); - } - return pn; -} */ - -void IvalueAnalysis::insert_shape_profile(torch::jit::Node* node, size_t offset) { - torch::jit::Value* input_value = node->input(offset); - //watch this value - auto pn = create_profile_node(nullptr, {input_value}); - auto pno = pn->addOutput(); - pn->ty_(torch::jit::attr::profiled_type, c10::TensorType::get()); - pno->setType(c10::TensorType::get()); - - std::function shape_profiler = [this, pno](torch::jit::Stack& stack) { - int64_t frame_id = 0; - torch::jit::pop(stack, frame_id); - c10::IValue ivalue; - torch::jit::pop(stack, ivalue); - if (ivalue.isTensor()) { - std::lock_guard lock(this->mutex_); - std::map>& profiled_types = _profiled_types_per_frame[frame_id]; - at::Tensor t = ivalue.toTensor(); - if (t.defined()) { - at::TensorTypePtr pttp = torch::jit::tensorTypeInCurrentExecutionContext(t); - if (profiled_types.count(pno) == 0) { - //insert value and tensortype info - profiled_types.insert({pno, {pttp}}); - } else { - std::vector& type_list = profiled_types.at(pno); - type_list.push_back(pttp); - // auto type = profiled_types.at(pno); - // pttp = type->merge(*pttp); - // profiled_types[pno] = pttp; - } - } else { - profiled_types[pno] = {c10::TensorType::get()->withUndefined()}; - } - } - // passing t through - torch::jit::push(stack, ivalue); - }; - pn->setCallback(shape_profiler); - pn->insertBefore(node); - node->replaceInput(offset, pn->output()); -} - -/* -void IvalueAnalysis::insert_optional_profile(torch::jit::Node* node, size_t offset) { - torch::jit::Value* input_value = node->input(offset); - // this value - auto opt_pn = create_profile_optional_node(nullptr, {input_value}); - // watch the definition instead of the use, - // because we are only optimizing in the case of a None value which is immutable - std::function optional_profiler = [this, opt_pn](torch::jit::Stack& stack) { - std::lock_guard lock(this->mutex_); - int64_t frame_id = 0; - torch::jit::pop(stack, frame_id); - c10::IValue ivalue; - torch::jit::pop(stack, ivalue); - if (ivalue.isNone()) { - opt_pn->i_(torch::jit::attr::num_none, opt_pn->i(torch::jit::attr::num_none) + 1); - } else { - opt_pn->i_(torch::jit::attr::num_present, opt_pn->i(torch::jit::attr::num_present) + 1); - } - torch::jit::push(stack, ivalue); - }; - opt_pn->setCallback(optional_profiler); - auto pno = opt_pn->addOutput(); - pno->setType(input_value->type()); - opt_pn->insertAfter(input_value->node()); - input_value->replaceAllUsesAfterNodeWith(opt_pn, pno); -} */ - -//TODO: check more -void IvalueAnalysis::insert_eval_profile(torch::jit::Node* node, size_t offset) { - torch::jit::Value* output_value = node->output(offset); - //watch this value - auto eval_pn = create_profile_node(nullptr, {output_value}); - auto pno = eval_pn->addOutput(); - eval_pn->ty_(torch::jit::attr::profiled_type, output_value->type()); - pno->setType(output_value->type()); - - //do we need outout? and change the input of prim::If? - std::function eval_profiler = [this, pno](torch::jit::Stack& stack) { - int64_t frame_id = 0; - torch::jit::pop(stack, frame_id); - c10::IValue ivalue; - torch::jit::pop(stack, ivalue); - std::lock_guard lock(this->mutex_); - if (ivalue.isBool()) { - //not exist yet, insert it - if (_evaluate_values_map.count(pno) == 0) { - std::vector bool_vector{ivalue.toBool()}; - _evaluate_values_map[pno] = bool_vector; - } else { - _evaluate_values_map[pno].emplace_back(ivalue.toBool()); - } - } - torch::jit::push(stack, ivalue); - }; - eval_pn->setCallback(eval_profiler); - eval_pn->insertAfter(node); - //replace the user nodes. - for (auto use: output_value->uses()) { - auto consumer_node = use.user; - for(size_t offset = 0; offset < consumer_node->inputs().size(); offset++) { - if (consumer_node->input(offset) == output_value && - consumer_node != eval_pn) { - consumer_node->replaceInput(offset, eval_pn->output()); - } - } - } -} - -std::map IvalueAnalysis::merge_tensor_type_per_frame( - std::map>& profiled_map) { - std::map merged_tensor_type; - for (auto iter : profiled_map) { - torch::jit::Value* profile_value = iter.first; - std::vector type_list = iter.second; - for (auto tensor_type : type_list) { - if (merged_tensor_type.count(profile_value) == 0) { - merged_tensor_type.insert({profile_value, tensor_type}); - } else { - c10::TensorTypePtr type = merged_tensor_type.at(profile_value); - tensor_type = type->merge(*tensor_type); - merged_tensor_type[profile_value] = tensor_type; - } - } - } - return merged_tensor_type; -} - -c10::SymbolicShape IvalueAnalysis::merge_symbolic_shapes( - const c10::SymbolicShape& new_sizes, - const c10::SymbolicShape& sym_shapes, - torch::jit::SetPartitioningHelper& partition_helper) { - std::vector new_symbols; - TORCH_INTERNAL_ASSERT( - new_sizes.rank().has_value() && sym_shapes.rank().has_value() && - *new_sizes.rank() == *sym_shapes.rank()); - - for (size_t i = 0; i < *new_sizes.rank(); i++) { - if (!(*sym_shapes.sizes())[i].is_static() || - !(*new_sizes.sizes())[i].is_static()) { - new_symbols.emplace_back(); - continue; - } - auto symbol = (*sym_shapes.sizes())[i]; - int64_t new_size = (*new_sizes.sizes())[i].static_size(); - //GRAPH_DUMP("Merging symbol ", symbol); - auto new_sym = partition_helper.partitionSetByDimension(new_size, symbol); - new_symbols.emplace_back(new_sym); - } - return c10::SymbolicShape(new_symbols); -} - -void IvalueAnalysis::analysis_ivalue_for_block(torch::jit::Block* block) { - for (auto it = block->nodes().begin(); it != block->nodes().end(); ++it) { - auto node = *it; - //iterate the input value of the node - for (size_t offset = 0; offset < node->inputs().size(); offset++) { - auto input_value = node->input(offset); - //tensortype handle - if (input_value->type()->kind() == c10::TypeKind::TensorType) { - insert_shape_profile(node, offset); - } - if (input_value->type()->kind() == c10::TypeKind::ListType){ - insert_input_listsize_profile(node, offset); - } - // 踩坑记录:0318,须保证后面分析的value没有被前面加过profile node - // 否则前面profile output替换了value节点,map key中找不到自己想要的value - // 且同一value的profile callback函数会多次执行,造成不可预知的问题 - if (input_value->type()->kind() == c10::TypeKind::IntType && - input_value->node()->kind() != torch::jit::prim::Constant) { - insert_number_eval_profile(node, offset); - } - //TODO: WHY NOT SUPPORT ProfileOptionalOp anymore after I upgrade libtorch from 1.7.1 to 1.8.1 - // if (input_value->type()->cast() && - // has_gradsum_to_size_uses(input_value)) { - // insert_optional_profile(node, offset); - // } - - if (input_value->type()->kind() == c10::TypeKind::BoolType) { - insert_eval_profile(input_value->node(), 0); - //TODO: modify the second input 0 to more strict check - } - } - for (size_t offset = 0; offset < node->outputs().size(); offset++) { - auto output_value = node->output(offset); - if (output_value->type()->kind() == c10::TypeKind::ListType){ - insert_output_listsize_profile(node, offset); - it++; - } - } - - for (auto b : node->blocks()) { - analysis_ivalue_for_block(b); - } - } - - //insert shape profile for block outputs - for (size_t offset = 0; offset < block->return_node()->inputs().size(); offset++) { - auto input_value = block->return_node()->input(offset); - if (input_value->type()->isSubtypeOf(c10::TensorType::get())) { - insert_shape_profile(block->return_node(), offset); - } - // //TODO: should I add this?? - // if (input_value->type()->kind() == c10::TypeKind::BoolType) { - // insert_eval_profile(input_value->node(), 0); - // } - } -} - -void IvalueAnalysis::gen_list_size() { - PorosGlobalContext::instance()._list_size_map = _list_size_map; -} - -void IvalueAnalysis::gen_value_dyanamic_shape() { - - std::map& value_dynamic_shape_map = PorosGlobalContext::instance()._value_dynamic_shape_map; - if (_profiled_types_per_frame.size() < 3) { - throw c10::Error("dynamic_shape must has three prewarm data [max & min & opt]", ""); - } - - auto profiled_types_iter = _profiled_types_per_frame.begin(); //frame id - auto start_frame_id = profiled_types_iter->first; // std::map - - //max - for (auto &e : _profiled_types_per_frame[start_frame_id++]) { - auto profile_value = e.first->node()->input(); - //没有则创建。 - if (value_dynamic_shape_map.count(profile_value) == 0) { - ValueDynamicShape shape; - value_dynamic_shape_map[profile_value] = shape; - value_dynamic_shape_map[profile_value].is_dynamic = false; - } - - std::vector& shape_list = e.second; - std::vector current_shape; - for (auto &shape : shape_list) { - if (shape->sizes().concrete_sizes().has_value()) { - current_shape = shape->sizes().concrete_sizes().value(); - } else { - // 因为此处在剪枝操作之前,有的block可能没有被执行。 - // 而这些block中的profile node是空值,此处应跳过 - continue; - } - //当一个value 作为多个node的输入的时候,会有多个profile。 - //其次,当一个tensor出现在loop中的时候,相关联的op大概率会被多次执行,也会有多个profile。 - //2021.11.11 踩坑记录,针对多个profile,如果出现了size 不一致的情况,max 应该取其中最大的,min应该取其中最小的。 - if (value_dynamic_shape_map[profile_value].max_shapes.size() != 0) { - auto old_shape = value_dynamic_shape_map[profile_value].max_shapes; - std::vector new_shape; - for (size_t i = 0; i < old_shape.size(); ++i) { - new_shape.push_back(std::max(old_shape[i], current_shape[i])); - } - value_dynamic_shape_map[profile_value].max_shapes = new_shape; - // LOG(INFO) << "try to update max shape, current_shape: [" << current_shape - // << "], old_shape: [" << old_shape - // << "], new_shape: [" << new_shape << "]"; - } else { - value_dynamic_shape_map[profile_value].max_shapes = current_shape; - } - } - } - - //min - for (auto &e : _profiled_types_per_frame[start_frame_id++]) { - //TODO: maybe need to check the value existing before setting - auto profile_value = e.first->node()->input(); - std::vector shape_list = e.second; - std::vector current_shape; - for (auto &shape : shape_list) { - if (shape->sizes().concrete_sizes().has_value()) { - current_shape = shape->sizes().concrete_sizes().value(); - } else { - // 因为此处在剪枝操作之前,有的block可能没有被执行。 - // 而这些block中的profile node是空值,此处应跳过 - continue; - } - if (value_dynamic_shape_map[profile_value].min_shapes.size() != 0) { - auto old_shape = value_dynamic_shape_map[profile_value].min_shapes; - std::vector new_shape; - for (size_t i = 0; i < old_shape.size(); ++i) { - new_shape.push_back(std::min(old_shape[i], current_shape[i])); - } - value_dynamic_shape_map[profile_value].min_shapes = new_shape; - // LOG(INFO) << "try to update min shape, current_shape: [" << current_shape - // << "], old_shape: [" << old_shape - // << "], new_shape: [" << new_shape << "]"; - } else { - value_dynamic_shape_map[profile_value].min_shapes = current_shape; - } - } - } - - //opt - for (auto &e : _profiled_types_per_frame[start_frame_id++]) { - auto profile_value = e.first->node()->input(); - std::vector shape_list = e.second; - for (auto &shape : shape_list) { - if (shape->sizes().concrete_sizes().has_value()) { - value_dynamic_shape_map[profile_value].opt_shapes = shape->sizes().concrete_sizes().value(); - } - } - } - - for (auto &e : value_dynamic_shape_map) { - ValueDynamicShape& shape = e.second; - //2022.09.28 踩坑记录,当针对某一个value, 出现了其中一个shape的size为0的情况, - //说明这个value在某个block下(可能是循环次数跟query相关的loop,也可能是进入条件跟query相关的if分支) - //此时对该block下的graph进行子图分割会出现异常,因为在子图替换阶段,无法正常生产输入的size信息。 - //此处兼容这种情况。 - if (shape.max_shapes.size() == 0 || shape.min_shapes.size() == 0 || shape.opt_shapes.size() == 0) { - if (e.first->node()->kind() == torch::jit::prim::Constant) { - continue; - } - LOG(INFO) << "value shape info for: %" << e.first->debugName() - << ", max_shape: " << shape.max_shapes - << ", min_shape: " << shape.min_shapes - << ", opt_shape: " << shape.opt_shapes; - PorosGlobalContext::instance()._disable_subblock_convert = true; - continue; - } - for (size_t i = 0; i < shape.max_shapes.size(); ++i) { - if (shape.max_shapes[i] == shape.min_shapes[i] && shape.max_shapes[i] == shape.opt_shapes[i]) { - shape.sizes.push_back(shape.max_shapes[i]); - } else { - shape.sizes.push_back(-1); - shape.is_dynamic = true; - } - } - // LOG(INFO) << "value shape info for: %" << e.first->debugName() - // << ", max_shape: " << shape.max_shapes - // << ", min_shape: " << shape.min_shapes - // << ", opt_shape: " << shape.opt_shapes; - } -} - -void IvalueAnalysis::gen_int_intlist_value() { - std::map& int_intlist_values_map = PorosGlobalContext::instance()._int_intlist_values_map; - if (_int_intlist_values_per_frame.size() == 0) { - return; - } - - int64_t start_frame_id = _int_intlist_values_per_frame.begin()->first; //frame id - - //max - // e -> std::map>> 迭代 pair - for (auto &e : _int_intlist_values_per_frame[start_frame_id++]) { - std::vector> int_values_vecs = e.second; - if (int_values_vecs.size() == 0) { - continue; - } - size_t per_vec_size = int_values_vecs[0].size(); - std::vector max_vector(per_vec_size, INT64_MIN); - bool length_is_var = false; - for (std::vector& v : int_values_vecs) { - if (max_vector.size() != v.size()) { - length_is_var = true; - break; - } - for (size_t i = 0; i < max_vector.size(); i++) { - max_vector[i] = std::max(max_vector[i], v[i]); - } - } - if (length_is_var) { - continue; - } - - torch::jit::Value* int_value = e.first; - if (int_intlist_values_map.count(int_value) == 0) { - ValueDynamicShape shape; - shape.max_shapes = max_vector; - int_intlist_values_map[int_value] = shape; - int_intlist_values_map[int_value].is_dynamic = false; - } else { - int_intlist_values_map[int_value].max_shapes = max_vector; - } - } - - if (_int_intlist_values_per_frame.size() == 1) { - for (auto &e : int_intlist_values_map) { - e.second.min_shapes = e.second.max_shapes; - e.second.opt_shapes = e.second.max_shapes; - } - return; - } else { - for (auto &e : _int_intlist_values_per_frame[start_frame_id++]) { - std::vector> int_values_vecs = e.second; - if (int_values_vecs.size() == 0) { - continue; - } - size_t per_vec_size = int_values_vecs[0].size(); - std::vector min_vector(per_vec_size, INT64_MAX); - bool length_is_var = false; - for (std::vector& v : int_values_vecs) { - if (min_vector.size() != v.size()) { - length_is_var = true; - break; - } - for (size_t i = 0; i < min_vector.size(); i++) { - min_vector[i] = std::min(min_vector[i], v[i]); - } - } - if (length_is_var) { - continue; - } - - torch::jit::Value* int_value = e.first; - if (int_intlist_values_map.count(int_value) == 0) { - ValueDynamicShape shape; - shape.min_shapes = min_vector; - int_intlist_values_map[int_value] = shape; - int_intlist_values_map[int_value].is_dynamic = false; - } else { - int_intlist_values_map[int_value].min_shapes = min_vector; - } - } - - //opt - for (auto &e : _int_intlist_values_per_frame[start_frame_id++]) { - std::vector> int_values_vecs = e.second; - if (int_values_vecs.size() == 0 ) { - continue; - } - size_t per_vec_size = int_values_vecs[0].size(); - bool length_is_var = false; - for (std::vector& v : int_values_vecs) { - if (per_vec_size != v.size()) { - length_is_var = true; - break; - } - } - if (length_is_var) { - continue; - } - - torch::jit::Value* int_value = e.first; - if (int_intlist_values_map.count(int_value) == 0) { - ValueDynamicShape shape; - shape.opt_shapes = int_values_vecs[0]; - int_intlist_values_map[int_value] = shape; - int_intlist_values_map[int_value].is_dynamic = false; - } else { - int_intlist_values_map[int_value].opt_shapes = int_values_vecs[0]; - } - } - } -} - -std::unique_ptr IvalueAnalysis::analysis_ivalue_for_graph( - const std::shared_ptr& graph) { - - auto new_g = graph->copy(); //copy or use the original one?? - auto ia = std::unique_ptr(new IvalueAnalysis(new_g)); - auto raw_ia = ia.get(); - - //clear the existing profile node that may exist. - torch::jit::ClearProfilingInformation(new_g); - //analysis main function - ia->analysis_ivalue_for_block(new_g->block()); - - std::function counter = [raw_ia](torch::jit::Stack& stack) { - int64_t frame_id = 0; - torch::jit::pop(stack, frame_id); - std::lock_guard lock(raw_ia->mutex_); - - if (raw_ia->profiling_count_ > 0) { - raw_ia->profiling_count_--; - } - - // merge tensortype profiling information from all runs - if (raw_ia->profiling_count_ == 0) { - LOG(INFO) << "Collected tensor profile " << raw_ia->_profiled_types_per_frame.size() << " records for run " << frame_id; - if (raw_ia->_profiled_types_per_frame.empty()) { - return; - } - // the key is a frame id, the value is a mapping from a Value in a graph to a profiled TensorType - // we make a copy of profiling information from the very first run - // and use it for building the symbol sets - auto profiled_types_iter = raw_ia->_profiled_types_per_frame.begin(); //frame id - - // merge itself - auto merged_profiled_types = raw_ia->merge_tensor_type_per_frame(profiled_types_iter->second); - ++profiled_types_iter; - - // merge profiling information from next runs into the first one - for (; profiled_types_iter != raw_ia->_profiled_types_per_frame.end(); ++profiled_types_iter) { - torch::jit::SetPartitioningHelper partition_helper; - for (const auto& val_type_pair : raw_ia->merge_tensor_type_per_frame(profiled_types_iter->second)) { - auto insertion_result = merged_profiled_types.insert(val_type_pair); - if (!insertion_result.second) { // Already existed - const c10::TensorType* type = insertion_result.first->second.get(); - //TODO: merge function take care more - //TODO: the merge function has change from torch1.7.1 to torch1.8.1 - auto merged_type = type->merge(*val_type_pair.second); - if (merged_type->sizes().size().has_value()) { - auto new_shape = raw_ia->merge_symbolic_shapes( - val_type_pair.second->symbolic_sizes(), type->symbolic_sizes(), partition_helper); - GRAPH_DEBUG("Merging ", *val_type_pair.second, " of run ", profiled_types_iter->first, " into ", *type); - merged_type = type->withSymbolicShapes(std::move(new_shape)); - GRAPH_DEBUG("Result : ", *merged_type); - insertion_result.first->second = std::move(merged_type); - } else { - // reset symbolic shapes when ranks are different - // TODO: attention here - insertion_result.first->second = std::move(merged_type); - } - } - } - } - - // update types in the graph - for (auto val_type_pair : merged_profiled_types) { - val_type_pair.first->node()->ty_(torch::jit::attr::profiled_type, val_type_pair.second); - } - } - - //TODO: check this more - // update eval information from all runs - if (raw_ia->profiling_count_ == 0) { - LOG(INFO) << "Collected evaluate " << raw_ia->_evaluate_values_map.size() << " records for run " << frame_id; - if (raw_ia->_evaluate_values_map.empty()) { - return; - } - - torch::jit::WithInsertPoint guard(raw_ia->profiled_graph_->block()->nodes().front()); - auto true_const = raw_ia->profiled_graph_->insertConstant(true); - auto false_const = raw_ia->profiled_graph_->insertConstant(false); - for (auto& value_bools_pair : raw_ia->_evaluate_values_map) { - auto profile_value = value_bools_pair.first; - auto bool_vector = value_bools_pair.second; - if (std::all_of(bool_vector.begin(), bool_vector.end(), - [](bool i){ return i == true;})) { - profile_value->node()->replaceInput(0, true_const); - //LOG(INFO) << "Replace " << node_info(profile_value->node()) << "input 0 as true_constant"; - } - - if (std::all_of(bool_vector.begin(), bool_vector.end(), - [](bool i){ return i == false;})) { - profile_value->node()->replaceInput(0, false_const); - //LOG(INFO) << "Replace " << node_info(profile_value->node()) << "input 0 as false_constant"; - } - } - } - }; //func counter end - - auto pop = ia->create_profile_node(counter, {}); - new_g->appendNode(pop); //put this profile at end of the graph to upback all the tensors. - GRAPH_DUMP("Instrumented Graph: ", new_g); - return ia; -} - -//DEPRECATED -bool has_gradsum_to_size_uses(torch::jit::Value* v) { - return std::any_of(v->uses().begin(), v->uses().end(), [](const torch::jit::Use& use) { - return use.user->kind() == torch::jit::aten::_grad_sum_to_size; - }); -} - -//DEPRECATED -void IvalueAnalysis::insert_debug_profile(torch::jit::Node* node, size_t offset) { - torch::jit::Value* input_value = node->input(offset); - //watch this value - auto pn = create_profile_node(nullptr, {input_value}); - //auto pno = pn->addOutput(); - pn->ty_(torch::jit::attr::profiled_type, c10::TensorType::get()); - //pno->setType(c10::TensorType::get()); - - std::function debug_profiler = [this, node](torch::jit::Stack& stack) { - int64_t frame_id = 0; - torch::jit::pop(stack, frame_id); - c10::IValue ivalue; - torch::jit::pop(stack, ivalue); - if (ivalue.isTensor()) { - std::lock_guard lock(this->mutex_); - auto t = ivalue.toTensor(); - if (t.defined()) { - auto pttp = torch::jit::tensorTypeInCurrentExecutionContext(t); - - //here. print node info. print input info - // std::cout << "debug during interprete, [node_info]:" << node_info_with_attr(node) - // <<", [input value type]: " << pttp->str() - // <<", [input value shape]: " << pttp->sizes().size() - // << std::endl; - } - } - // passing t through - torch::jit::push(stack, ivalue); - }; - - pn->setCallback(debug_profiler); - pn->insertBefore(node); - //node->replaceInput(offset, pn->output()); -} - -//DEPRECATED -void IvalueAnalysis::debug_tensors_for_block(torch::jit::Block* block) { - for (auto it = block->nodes().begin(); it != block->nodes().end(); ++it) { - auto node = *it; - //iterate the input value of the node - for (size_t offset = 0; offset < node->inputs().size(); offset++) { - auto input_value = node->input(offset); - //tensortype handle - if (input_value->type()->kind() == c10::TypeKind::TensorType) { - insert_debug_profile(node, offset); - } - } - - for (auto b : node->blocks()) { - debug_tensors_for_block(b); - } - } - //insert shape profile for block outputs - for (size_t offset = 0; offset < block->return_node()->inputs().size(); offset++) { - auto input_value = block->return_node()->input(offset); - if (input_value->type()->isSubtypeOf(c10::TensorType::get())) { - insert_debug_profile(block->return_node(), offset); - } - } -} - -//DEPRECATED -std::vector get_prim_if_user(torch::jit::Value* value) { - std::vector if_nodes; - for (auto use : value->uses()) { - if (is_dim_equal_if_node(use.user)) { - if_nodes.emplace_back(use.user); - } - } - //sort - std::sort(if_nodes.begin(), if_nodes.end(), [&](torch::jit::Node* a, torch::jit::Node* b) { - return a->isBefore(b); - }); - return if_nodes; -} - -//DEPRECATED -void IvalueAnalysis::prune_if_block(torch::jit::Block* block) { - if (_evaluate_values_map.empty()) { - return; - } - for (auto itr = block->nodes().begin(); itr != block->nodes().end(); itr++) { - auto node = *itr; - //itr++; // nonono, not here. the next node may be if node. and may be already destroyed below - if (node->kind() == torch::jit::prim::profile && - node->outputs().size() == 0 && node->inputs().size() == 1) { - auto evaluate_values = _evaluate_values_map.find(node->input(0)); - if (evaluate_values != _evaluate_values_map.end()) { - auto bool_vector = evaluate_values->second; - if (std::all_of(bool_vector.begin(), bool_vector.end(), - [](bool i){ return i == true;})) { - //the result keep true during every data round - auto if_nodes = get_prim_if_user(node->input(0)); - for (auto if_node: if_nodes) { - inline_if_body(if_node->blocks().at(0)); - } - } - if (std::all_of(bool_vector.begin(), bool_vector.end(), - [](bool i){ return i == false;})) { - //the result keep false during every round - auto if_nodes = get_prim_if_user(node->input(1)); - for (auto if_node: if_nodes) { - inline_if_body(if_node->blocks().at(1)); - } - } - } - //cause it has no output. so destroy it directly. - node->destroy(); - _evaluate_values_map.erase(node->input(0)); - } else { - for (torch::jit::Block* ib : node->blocks()) { - prune_if_block(ib); - } - } - } -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/compile/ivalues_analysis.h b/poros/poros/compile/ivalues_analysis.h deleted file mode 100644 index 69be38ecc5e..00000000000 --- a/poros/poros/compile/ivalues_analysis.h +++ /dev/null @@ -1,119 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file ivalue_analysis.h -* @author tianjinjin@baidu.com -* @date Thu Mar 18 14:33:54 CST 2021 -* @brief -**/ - -#pragma once - -#include -#include -#include -#include - -#include -#include -#include -#include //for Stack -#include //for ProfileOp -#include //for SetPartitioningHelper - -#include "poros/context/poros_global.h" - -namespace baidu { -namespace mirana { -namespace poros { - -struct IvalueAnalysis { - //disable copy and move op to avoid unexpected copy/move happened when in callback func - IvalueAnalysis(const IvalueAnalysis&) = delete; - IvalueAnalysis(IvalueAnalysis&&) noexcept = delete; - static std::unique_ptr analysis_ivalue_for_graph( - const std::shared_ptr& graph); - - std::shared_ptr profiled_graph_; - std::mutex mutex_; - size_t profiling_count_; - // the key is a frame id - // the value is a mapping from a Value in a graph to a profiled TensorType - std::map>> _profiled_types_per_frame; - // meaning of key(int64_t) and value(Value*) are same to _profiled_types_per_frame. - // vec records all of int(int[]) values against the Value* in a graph. - std::map>>> _int_intlist_values_per_frame; - // std::map> _evaluate_values_per_frame; - // we only store bool data. this may change in the future. - std::map> _evaluate_values_map; - - // 存储list类型的value相关信息 - ListSizeMap _list_size_map; - - std::shared_ptr graph() const { - return profiled_graph_; - } - - // 拷贝dynamic信息到context - void gen_value_dyanamic_shape(); - - // 拷贝list信息到context - void gen_list_size(); - - // 拷贝int int[]值信息到context - void gen_int_intlist_value(); - - private: - //ProfileIValueOp not supported when in pytorch 1.7.x - //so I have to rollback to ProfileOp, and have to copy the main function - torch::jit::ProfileOp* create_profile_node( - const std::function& fp, - at::ArrayRef inputs); - - void analysis_ivalue_for_block(torch::jit::Block* block); - void insert_shape_profile(torch::jit::Node* node, size_t offset); - void insert_eval_profile(torch::jit::Node* node, size_t offset); - void insert_input_listsize_profile(torch::jit::Node* node, size_t offset); - void insert_output_listsize_profile(torch::jit::Node* node, size_t offset); - void insert_number_eval_profile(torch::jit::Node* node, size_t offset); - - /** - * merge the tensortype list about one given value to a single merged tensortype - * **/ - std::map merge_tensor_type_per_frame( - std::map>& profiled_map); - - c10::SymbolicShape merge_symbolic_shapes( - const c10::SymbolicShape& new_sizes, - const c10::SymbolicShape& sym_shapes, - torch::jit::SetPartitioningHelper& partition_helper); - - //DEPRECATED - //cut down some if block if the if-condition won't change during all the warm-up data - void prune_if_block(torch::jit::Block* block); - //DEPRECATED - void debug_tensors_for_block(torch::jit::Block* block); - //DEPRECATED - void insert_debug_profile(torch::jit::Node* node, size_t offset); - - - //be private. - IvalueAnalysis(std::shared_ptr g); -}; - - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/compile/partition.cpp b/poros/poros/compile/partition.cpp deleted file mode 100644 index 401da922e35..00000000000 --- a/poros/poros/compile/partition.cpp +++ /dev/null @@ -1,192 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file partition.cpp -* @author tianjinjin@baidu.com -* @date Thu Jun 3 15:10:30 CST 2021 -* @brief -**/ - -#include "poros/compile/partition.h" - -namespace baidu { -namespace mirana { -namespace poros { - -// utility function to check if the node implies broadcast on a given shape ( -// assumed to be shape of an input tensor) -// limitations: -// 1. we rely on shape information to judge this. so we would require output -// shape to be available; -// 2. we basically compares given shape to the shape of the only output of -// the node and return true if it implies broadcast from the former to the -// latter. -bool maybeBroadcastOnShape( - const torch::jit::Node* node, - const std::vector>& shape) { - //TODO: add outputs size check - //TORCH_INTERNAL_ASSERT(n->outputs().size() == 1, "not expecting multiple outputs from a node, graph partitioning logic needs to be updated"); - // assumes that if output is not a tensor type, it's not broadcasting - if (auto out_type = node->output(0)->type()->cast()) { - if (out_type->dim()) { - if (out_type->dim().value() < shape.size()) { - // no broadcast for reduction operation; - return false; - } else if (out_type->dim().value() > shape.size()) { - // increased rank means there is reduction; - return true; - } else { - // same rank, we need to iterate through sizes and check if size-1 - // exists in input `shape` - for (const auto& opt_size : shape) { - // TODO: not sure if we need to check for output size != 1, since we - // are currently marking all size-1 dimension as broadcast in codegen. - if (opt_size.has_value() && opt_size.value() == 1) { - return true; - } - } - } - } - } - return false; -}; - -// bool hasReductionOperation(const torch::jit::Node* node) { -// if (torch::jit::fuser::cuda::isReductionNode(node)) { -// return true; -// } -// if (node->kind() == torch::jit::prim::CudaFusionGroup) { -// for (auto n : node->g(torch::jit::attr::Subgraph)->nodes()) { -// if (hasReductionOperation(n)) { -// return true; -// } -// } -// } -// return false; -// } - -bool createTrickyBroadcast(const torch::jit::Node* consumer, const torch::jit::Node* producer) { - - auto count_broadcasting_in_node = - [](const torch::jit::Node* node, - const std::vector>& shape, - size_t offset) { - int num_broadcasting = 0; - if (node->kind() == torch::jit::prim::CudaFusionGroup) { - // be careful here as `subgraph_input`, as its name suggests, is in a - // different fraph from `node`. - const auto& subgraph_input =node->g(torch::jit::attr::Subgraph)->inputs()[offset]; - for (const auto& use : subgraph_input->uses()) { - if (maybeBroadcastOnShape(use.user, shape)) { - num_broadcasting++; - } - } - } else { - if (maybeBroadcastOnShape(node, shape)) { - num_broadcasting++; - } - } - return num_broadcasting; - }; - - // case 1. We check shared inputs to `producer` & `consumer`; - for (int i = 0; i < static_cast(producer->inputs().size()); i++) { - auto n_input = producer->input(i); - auto n_input_type = n_input->type()->cast(); - if (n_input_type != nullptr && n_input_type->sizes().sizes()) { - std::vector> n_input_shape = n_input_type->sizes().sizes().value(); - int num_broadcasting = 0; - - // check broadcasting for the n_input inside `consumer`; - for (const auto& use : n_input->uses()) { - if (use.user == consumer) { - num_broadcasting += count_broadcasting_in_node(consumer, n_input_shape, use.offset); - } - } - - // if no broadcasting happened for consumer, there's no point check - // multiple broadcasting in producer alone; - if (num_broadcasting == 0) { - continue; - } - - // check broadcasting for n_input inside `producer`; - num_broadcasting += count_broadcasting_in_node(producer, n_input_shape, i); - - // encounted multiple broadcasting scheme for a single TV, we will not be - // able to schedule this, prevent the fusion; (case 1) - if (num_broadcasting > 1) { - return true; - } - } - } - - // case 2. We check input to `consumer` that is also the output from - // `producer` - for (int i = 0; i < static_cast(producer->outputs().size()); i++) { - auto n_output = producer->output(i); - auto n_output_type = n_output->type()->cast(); - if (n_output_type != nullptr && n_output_type->sizes().sizes()) { - std::vector> n_output_shape = n_output_type->sizes().sizes().value(); - int num_broadcasting = 0; - // If we only look at case 1 & case 2, we need to check broadcast of - // `n_output` inside `producer`, if it is a `prim::CudaFusionGroup`. - // this is actually not necessary when we consider case 3, as we avoid - // broadcasting on outputs already; - - // TODO: merge this code with case 1. - // check broadcasting for the n_output inside `consumer`; - bool use_as_output = false; - for (const auto& use : n_output->uses()) { - if (use.user == consumer) { - num_broadcasting += count_broadcasting_in_node(consumer, n_output_shape, use.offset); - } else { - // case 3. output is used by other nodes not the consumer, no - // broadcasting is allowed; - use_as_output = true; - } - } - - // encounted multiple broadcasting scheme for a single TV, we will not be - // able to schedule this, prevent the fusion; (case 2) - // Alternatively, if use_as_output is true, we would not permit broadcast - // at all. (case 3) - if (num_broadcasting > (use_as_output ? 0 : 1)) { - return true; - } - } - } - return false; -} - -bool is_node_fusable(const torch::jit::Node* node, IEngine* engine) { - if (node->kind() == torch::jit::prim::CudaFusionGroup || (engine->is_node_supported(node))) { - return true; - } - return false; -} - -bool is_node_fusable(const torch::jit::Node* fusion, - const torch::jit::Node* node, - IEngine* engine) { - if (is_node_fusable(node, engine) && !createTrickyBroadcast(fusion, node)) { - return true; - } - return false; -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/compile/partition.h b/poros/poros/compile/partition.h deleted file mode 100644 index 3f193197c84..00000000000 --- a/poros/poros/compile/partition.h +++ /dev/null @@ -1,39 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file partition.h -* @author tianjinjin@baidu.com -* @date Thu Jun 3 14:57:58 CST 2021 -* @brief -**/ - -#pragma once - -#include "torch/script.h" - -#include "poros/engine/iengine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -bool is_node_fusable(const torch::jit::Node* node, IEngine* engine); -bool is_node_fusable(const torch::jit::Node* fusion, - const torch::jit::Node* node, - IEngine* engine); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/compile/poros_module.cpp b/poros/poros/compile/poros_module.cpp deleted file mode 100644 index 5b4e618a532..00000000000 --- a/poros/poros/compile/poros_module.cpp +++ /dev/null @@ -1,53 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file poros_module.cpp -* @author huangben@baidu.com -* @date 2021/08/05 11:39:03 CST 2021 -* @brief -**/ - -#include "poros/compile/poros_module.h" - -namespace baidu { -namespace mirana { -namespace poros { - -std::unique_ptr Load(const std::string& filename, const PorosOptions& options) { - torch::jit::Module module; - try { - module = torch::jit::load(filename); - } catch (const c10::Error& e) { - LOG(ERROR) << "error loading the model"; - return nullptr; - } - std::unique_ptr poros_module(new PorosModule(module)); - poros_module->_options = options; - - if (options.device == GPU) { - poros_module->to(at::kCUDA); - } - - if (options.debug == true) { - // when setting this, all the INFO level will be printed - c10::ShowLogInfoToStderr(); - } - - return poros_module; -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/compile/poros_module.h b/poros/poros/compile/poros_module.h deleted file mode 100644 index 9a836b437a2..00000000000 --- a/poros/poros/compile/poros_module.h +++ /dev/null @@ -1,89 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file poros_module.h -* @author huangben@baidu.com -* @date 2021/08/05 5 11:39:03 CST 2021 -* @brief -**/ - -#pragma once - -#include -#include -#include -//#include - -namespace baidu { -namespace mirana { -namespace poros { - -enum Device : int8_t { - GPU = 0, - CPU, - XPU, - UNKNOW -}; - -struct PorosOptions { - Device device = GPU; - bool debug = false; - bool use_fp16 = false; - bool is_dynamic = false; - // 该flag对tensorrt engine 有效, 默认为true - // 当long_to_int=true,将相关value转成at::kInt进行处理(因为tensorrt不支持at::kLong 类型) - // 该设置可能导致数据精度发生变化,如果效果不符合预期,请将该flag设置为false。 - bool long_to_int = true; - //DynamicShapeOptions dynamic_shape_options; - uint64_t max_workspace_size = 1ULL << 30; - // XPU默认参数为-1,代表第一个可用设备 - int32_t device_id = -1; - // 非const op个数阈值 - int32_t unconst_ops_thres = -1; - // Nvidia TF32 computes inner products by rounding the inputs to 10-bit mantissas before multiplying, - // but accumulates the sum using 23-bit mantissas to accelerate the calculation. - // note: It will work on ampere architecture (such as: A10), but may cause diff to the results. - bool use_nvidia_tf32 = true; - // preprocess mode - // 0: use torch.jit.script - // 1: use torhc.jit.trace - int32_t preprocess_mode = 0; - // 用户自定义不支持op列表 - std::vector unsupport_op_list; -}; - -class PorosModule : public torch::jit::Module { -public: - PorosModule(torch::jit::Module module) : torch::jit::Module(module) { - } - ~PorosModule() = default; - - void to_device(Device device){ - _options.device = device; - } - - //c10::IValue forward(std::vector inputs); - //void save(const std::string& filename); -public: - PorosOptions _options; - -}; - -//via porosmodule.save -std::unique_ptr Load(const std::string& filename, const PorosOptions& options); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/compile/segment.cpp b/poros/poros/compile/segment.cpp deleted file mode 100644 index 0bf88e58079..00000000000 --- a/poros/poros/compile/segment.cpp +++ /dev/null @@ -1,231 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file segment.cpp -* @author tianjinjin@baidu.com -* @date Fri Mar 19 19:18:20 CST 2021 -* @brief -**/ - -#include "poros/compile/segment.h" - -namespace baidu { -namespace mirana { -namespace poros { - -std::vector sort_topological(const at::ArrayRef inputs, - const torch::jit::Block* cur_block, - bool reverse) { - //if not in the same block. bypass it - std::vector result; - for (auto i : inputs) { - if (i->node()->owningBlock() == cur_block) { - result.push_back(i); - } - } - - if (reverse) { - // Sort in reverse topological order - std::sort(result.begin(), result.end(), [&](const torch::jit::Value* a, const torch::jit::Value* b) { - return a->node()->isAfter(b->node()); - }); - } else { - std::sort(result.begin(), result.end(), [&](const torch::jit::Value* a, const torch::jit::Value* b) { - return a->node()->isBefore(b->node()); - }); - } - return result; -} - -std::vector sort_topological (const at::ArrayRef inputs, - const torch::jit::Block* cur_block, - bool reverse) { - //if not in the same block. bypass it - std::vector result; - for (auto i : inputs) { - if (i->node()->owningBlock() == cur_block) { - result.push_back(i); - } - } - - if (reverse) { - // Sort in reverse topological order - std::sort(result.begin(), result.end(), [&](torch::jit::Value* a, torch::jit::Value* b) { - return a->node()->isAfter(b->node()); - }); - } else { - std::sort(result.begin(), result.end(), [&](torch::jit::Value* a, torch::jit::Value* b) { - return a->node()->isBefore(b->node()); - }); - } - return result; -} - -void stable_dfs(const torch::jit::Block& block, bool reverse, - const std::vector& start, - const std::function& enter, - const std::function& leave) -{ - std::vector stack(start.size()); - for (size_t i = 0; i < start.size(); ++i) { - stack[i] = NodeDFSResult{start[i], false}; - } - - std::unordered_map visited; - while(!stack.empty()) { - NodeDFSResult w = stack.back(); - stack.pop_back(); - - auto n = w.node; - if (w.leave) { - if (leave && !leave(n)) { - return; - } - continue; - } - - if (visited.find(n) != visited.end()) { - continue; - } - visited[n] = true; - - if (enter && !enter(n)) { - return; - } - - if (leave) { - stack.push_back(NodeDFSResult{n, true}); - } - - auto values = reverse ? n->inputs() : n->outputs(); - auto sorted_value_list = sort_topological(values, n->owningBlock(), false); - for (auto value: sorted_value_list) { - if (visited.find(value->node()) == visited.end()) { - stack.push_back(NodeDFSResult{n, false}); - } - } - } -} - -bool can_contract(const torch::jit::Node* from_node, - const torch::jit::Node* to_node, - const torch::jit::Block& block) { - std::vector dfs_start_nodes; - - for (auto i: to_node->inputs()) { - if (i->node() != from_node) { - dfs_start_nodes.push_back(i->node()); - } - } - - bool has_cycle = false; - stable_dfs (block, /*reverse=*/true, dfs_start_nodes, /*enter=*/nullptr, - [&has_cycle, from_node](const torch::jit::Node* n) { - if (n == from_node) { - has_cycle = true; - return false; - } - return true; - }); - return !has_cycle; -} - -torch::jit::Graph& get_subgraph(torch::jit::Node* n) { - AT_ASSERT(n->kind() == torch::jit::prim::CudaFusionGroup); - return *n->g(torch::jit::attr::Subgraph); - } - -torch::jit::Node* merge_node_into_subgraph(torch::jit::Node* group, torch::jit::Node* n) { - auto& subgraph = get_subgraph(group); - std::unordered_map inputs_map; - size_t i = 0; - size_t tensor_insert_idx = 0; - //cache the original group input data - AT_ASSERT(group->inputs().size() == subgraph.inputs().size()); - for (auto input : group->inputs()) { - inputs_map[input] = subgraph.inputs()[i++]; - if (input->type()->isSubtypeOf(c10::TensorType::get())) { - tensor_insert_idx = i; - } - } - - torch::jit::WithInsertPoint guard(*subgraph.nodes().begin()); - for (auto input : n->inputs()) { - //means we should add this new input - if (inputs_map.count(input) == 0) { - //consider tensortype first. (it's pytorch tradition) - if (input->type()->isSubtypeOf(c10::TensorType::get())) { - auto in_group = subgraph.insertInput(tensor_insert_idx); - in_group->setType(input->type()); - inputs_map[input] = in_group; - group->insertInput(tensor_insert_idx, input); - tensor_insert_idx++; - } else if ((input->type()->isSubtypeOf(c10::FloatType::get()) && - input->node()->kind() != torch::jit::prim::Constant) || - (n->kind() == torch::jit::aten::_grad_sum_to_size && - input->type()->isSubtypeOf(c10::ListType::ofInts()))) { - auto in_group = subgraph.addInput(); - in_group->setType(input->type()); - inputs_map[input] = in_group; - group->addInput(input); - } else if (input->node()->kind() == torch::jit::prim::Constant) { - torch::jit::Node* in_const = subgraph.createClone(input->node(), [](torch::jit::Value*) -> torch::jit::Value* { - throw std::runtime_error("unexpected input"); - }); - subgraph.insertNode(in_const); - inputs_map[input] = in_const->output(); - } else { - // TODO: we need to figure out what are supported input scalar - LOG(WARNING) << "meet some unexpected node: " << input->node()->kind().toQualString(); - auto in_group = subgraph.addInput(); - in_group->setType(input->type()); - inputs_map[input] = in_group; - group->addInput(input); - } - } - } // for (auto input : n->inputs()) - - // copy n into the graph, remapping its inputs to internal nodes - torch::jit::Node* in_graph = subgraph.createClone( - n, [&](torch::jit::Value* k) -> torch::jit::Value* { return inputs_map[k]; }); - - auto inputs = group->inputs(); - for (size_t i = 0; i < n->outputs().size(); ++i) { - auto it = std::find(inputs.begin(), inputs.end(), n->outputs()[i]); - if (it != inputs.end()) { - size_t p = it - inputs.begin(); - group->removeInput(p); - subgraph.inputs()[p]->replaceAllUsesWith(in_graph->outputs()[i]); - subgraph.eraseInput(p); - } - } - return subgraph.insertNode(in_graph); -} - -torch::jit::Node* change_node_to_subgraph(torch::jit::Node* group, torch::jit::Node* n) -{ - group->insertBefore(n); - torch::jit::Node* mergedNode = merge_node_into_subgraph(group, n); - get_subgraph(group).registerOutput(mergedNode->output()); - auto sel = group->addOutput(); - sel->copyMetadata(n->output()); - n->replaceAllUsesWith(group); - n->destroy(); - return group; -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/compile/segment.h b/poros/poros/compile/segment.h deleted file mode 100644 index 5eb1a8e344e..00000000000 --- a/poros/poros/compile/segment.h +++ /dev/null @@ -1,129 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file segment.h -* @author tianjinjin@baidu.com -* @date Thu Mar 18 14:33:54 CST 2021 -* @brief -**/ - -#pragma once - -#include -#include -#include -#include -#include - -#include "poros/engine/iengine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -struct NodeDFSResult { - const torch::jit::Node* node; - bool leave; // Are we entering or leaving n? -}; - -//template //should I? -class NodeLink { -public: - NodeLink(torch::jit::Node* n) - : _size(1), _head(nullptr), _value(n) {} - - ~NodeLink() { - if (_head != nullptr) { - _head = nullptr; - } - if (_value != nullptr) { - _value = nullptr; - } - } - - int size() {return find_root()->_size; } - int merge(NodeLink* other) { - NodeLink* a = find_root(); - NodeLink* b = other->find_root(); - if (a == b) { - return 0; - } - b->_head = a; - a->_size += b->_size; - return 0; - }; - - // Retrieves the value for the root of the set. - torch::jit::Node* head_value() { return find_root()->_value; } - - // Returns the value for the object. - torch::jit::Node* value() const { return _value; } - //int64_t value_index() {return _value->topo_position_; } - -private: - NodeLink* find_root() { - if (!_head) { - return this; - } - _head = _head->find_root(); - return _head; - }; - - int _size; - NodeLink* _head; - torch::jit::Node* _value; -}; - - -struct SegmentOptions { - int minimum_segment_size = 2; //每个setment至少包含多少个node。 -}; - -struct Segment { - Segment() {} - Segment(std::set& nodes) - : nodes(nodes){} - std::set nodes; -}; - -using SegmentVector = std::vector; -using ValueVector = std::vector; - -ValueVector sort_topological (const at::ArrayRef inputs, - const torch::jit::Block* cur_block, - bool reverse = false); -std::vector sort_topological (const at::ArrayRef inputs, - const torch::jit::Block* cur_block, - bool reverse = false); - -void stable_dfs(const torch::jit::Block& block, bool reverse, - const std::vector& start, - const std::function& enter, - const std::function& leave); - - -bool can_contract(const torch::jit::Node* from_node, - const torch::jit::Node* to_node, - const torch::jit::Block& block); - -torch::jit::Graph& get_subgraph(torch::jit::Node* n); -torch::jit::Node* merge_node_into_subgraph(torch::jit::Node* group, torch::jit::Node* n); -torch::jit::Node* change_node_to_subgraph(torch::jit::Node* group, torch::jit::Node* n); - -//void segment_graph_new(std::shared_ptr& graph, IEngine* engine); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/context/poros_global.cpp b/poros/poros/context/poros_global.cpp deleted file mode 100644 index 461d28158b3..00000000000 --- a/poros/poros/context/poros_global.cpp +++ /dev/null @@ -1,122 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file poros_global.cpp -* @author tianshaoqing@baidu.com -* @author huangben@baidu.com -* @date Fri Jul 23 11:21:10 CST 2021 -* @brief -**/ - -#include "poros/context/poros_global.h" -#include "poros/converter/iconverter.h" - -namespace baidu { -namespace mirana { -namespace poros { - -void ListSizeMap::update_value(torch::jit::Value* old_value, torch::jit::Value* new_value) { - if (_list_size_map_input.count(old_value) != 0) { - _list_size_map_input[new_value] = _list_size_map_input[old_value]; - _list_tensor_type_map_input[new_value] = _list_tensor_type_map_input[old_value]; - } - if (_list_size_map_output.count(old_value) != 0) { - _list_size_map_output[new_value] = _list_size_map_output[old_value]; - _list_tensor_type_map_output[new_value] = _list_tensor_type_map_output[old_value]; - } -} - -void ListSizeMap::update_node(torch::jit::Node* old_node, torch::jit::Node* new_node) { - for(size_t i = 0; i < new_node->inputs().size(); i++) { - auto value = new_node->input(i); - if (_list_size_map_input.count(value) != 0) { - if (_list_size_map_input[value].count(old_node) != 0) { - _list_size_map_input[value][new_node] = _list_size_map_input[value][old_node]; - _list_size_map_input[value].erase(old_node); - - _list_tensor_type_map_input[value][new_node] = _list_tensor_type_map_input[value][old_node]; - _list_tensor_type_map_input[value].erase(old_node); - } - } - } - - for(size_t i = 0; i < new_node->outputs().size(); i++) { - auto value = new_node->output(i); - if (_list_size_map_output.count(value) != 0) { - if (_list_size_map_output[value].count(old_node) != 0) { - _list_size_map_output[value][new_node] = _list_size_map_output[value][old_node]; - _list_size_map_output[value].erase(old_node); - - _list_tensor_type_map_output[value][new_node] = _list_tensor_type_map_output[value][old_node]; - _list_tensor_type_map_output[value].erase(old_node); - } - } - } -} - -// 将PorosOptions放到全局类中,之后再初始化用户自定义不支持op列表 -void PorosGlobalContext::set_poros_options(const PorosOptions& options) { - _poros_options = options; - for (auto i : _converters_map) { - i.second->init_unsupport_op_set(); - } -} - -// 注册converter方法到全局的PorosGlobalContext。 -void PorosGlobalContext::register_converter(const std::string& engine_name, IConverter* converter) { - //根据engine_name 找到相应的 ConvertersMap - //(ps: 不同engine的ConvertersMap是相互独立的) - auto search = _converters_map.find(engine_name); - if (search == _converters_map.end()) { - _converters_map[engine_name] = new ConvertersMap(); - } - auto e_converter_map = _converters_map[engine_name]; - - //根据converter的node_kind()和schema_string()信息,构造ConvRegistration。 - auto node_kind_list = converter->node_kind(); - auto schema_list = converter->schema_string(); - for (auto& node_kind : node_kind_list) { - //converter that without schemas. such as aten::Constant - ConvRegistration conv_reg; - conv_reg.kind = node_kind; - conv_reg.converter = converter; - if (schema_list.size() == 0) { - conv_reg.options = ConverterOptions(); - } else { - conv_reg.options = ConverterOptions().set_valid_schemas(schema_list); - } - //调用ConvertersMap的add_converter方法, 完成注册。 - e_converter_map->add_converter(node_kind, conv_reg); - } - return; -}; - -ConvertersMap* PorosGlobalContext::get_converter_map(const std::string& engine_name) { - auto search = _converters_map.find(engine_name); - if (search == _converters_map.end()) { - return nullptr; - } - return search->second; -}; - -void PorosGlobalContext::destroy() { - for (auto &e : _converters_map) { - delete e.second; - } -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/context/poros_global.h b/poros/poros/context/poros_global.h deleted file mode 100644 index 7d367fcb48a..00000000000 --- a/poros/poros/context/poros_global.h +++ /dev/null @@ -1,162 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file poros_global.h -* @author tianjinjin@baidu.com -* @author huangben@baidu.com -* @date Fri Jul 23 11:21:10 CST 2021 -* @brief -**/ - -#pragma once - -#include -#include -#include - -#include - -#include "poros/compile/poros_module.h" -#include "poros/iplugin/plugin_create.h" -#include "poros/util/macros.h" - -namespace baidu { -namespace mirana { -namespace poros { - -// 设置双层map的原因是为了解决在同一个value不同node输入list时,因为append导致的引用问题 -typedef std::map>> LIST_SIZE_MAP; -typedef std::map>>> TENSOR_LIST_TYPE_MAP; - -struct ListSizeMap { - // 存储输入输出为list类型时的size信息 - LIST_SIZE_MAP _list_size_map_input; - LIST_SIZE_MAP _list_size_map_output; - - // 存储输入输出类型为tensor list时的type信息 - TENSOR_LIST_TYPE_MAP _list_tensor_type_map_input; - TENSOR_LIST_TYPE_MAP _list_tensor_type_map_output; - - /** - * @brief 将old_value的信息更新到new_value上 - * @param [in] old_value : 原value - * @param [in] new_value : 新的value - * @return null - **/ - void update_value(torch::jit::Value* old_value, torch::jit::Value* new_value); - - /** - * @brief 将old_node的信息更新到new_node上 - * @param [in] old_node : 原node - * @param [in] new_node : 新的node - * @return null - **/ - void update_node(torch::jit::Node* old_node, torch::jit::Node* new_node); -}; - -struct ValueDynamicShape { - std::vector sizes; - std::vector max_shapes; - std::vector min_shapes; - std::vector opt_shapes; - bool is_dynamic = false; -}; - -// 前置声明 -class ConvertersMap; -class IConverter; - -class PorosGlobalContext { -public: - static PorosGlobalContext& instance() { - static PorosGlobalContext _instance; - return _instance; - } - - ~PorosGlobalContext() { - destroy(); - } - - int init() { - //to change - return 0; - } - - void set_poros_options(const PorosOptions& options); - - PorosOptions& get_poros_options() { - return _poros_options; - } - - void destroy(); - - // 注册converter方法到全局的PorosGlobalContext。 - void register_converter(const std::string& engine_name, IConverter* converter); - - ConvertersMap* get_converter_map(const std::string& engine_name); -public: - plugin_creator_map_t _engine_creator_map; - std::map _value_dynamic_shape_map; - ListSizeMap _list_size_map; - std::map _int_intlist_values_map; - bool _disable_subblock_convert = false; - - const std::set supported_mutable_ops_set = { - //aten::append.t(t[](a!) self, t(c -> *) el) -> t[](a!) - c10::Symbol::fromQualString("aten::append"), - //"aten::_set_item.t(t [](a!) l, int idx, t(b -> *) el) -> t[](a!)" - c10::Symbol::fromQualString("aten::_set_item"), - }; -private: - PorosOptions _poros_options; - std::unordered_map _converters_map; -}; - -/*------------------------------------------------------------------------- - converter自动注册相关宏 --------------------------------------------------------------------------*/ -template -class ConverterRegister { -public: - public: - inline ConverterRegister(std::string name = "", - PorosGlobalContext& context = PorosGlobalContext::instance()) noexcept; -}; - -template -inline ConverterRegister::ConverterRegister(std::string name, - PorosGlobalContext& context) noexcept { - auto instance = new T(); - context.register_converter(name, instance); -} - -#define POROS_REGISTER_CONVERTER(name, reg) \ - static ConverterRegister POROS_CONVERTER_REGISTER_init_ ## reg (#name); - -//engine自动注册 -template -class EngineRegister { -public: - EngineRegister(const std::string& name) { - register_plugin_class(name, PorosGlobalContext::instance()._engine_creator_map); - } -}; - -#define POROS_REGISTER_ENGINE(name) \ - static EngineRegister POROS_ENGINE_REGISTER_init_##name(#name); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/activation.cpp b/poros/poros/converter/gpu/activation.cpp deleted file mode 100644 index b55ed4065f5..00000000000 --- a/poros/poros/converter/gpu/activation.cpp +++ /dev/null @@ -1,234 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file activation.cpp -* @author tianjinjin@baidu.com -* @date Mon Mar 8 11:36:11 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/activation.h" -#include "poros/converter/gpu/weight.h" -#include "poros/util/macros.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/context/poros_global.h" -#include "poros/util/poros_util.h" -#include "poros/converter/gpu/converter_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -bool ActivationConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 1 || inputs.size() == 2 - || inputs.size() == 3 || inputs.size() == 4), - "invaid inputs size for ActivationConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for ActivationConverter is not Tensor as expected"); - - auto nv_tensor = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((nv_tensor != nullptr), - "Unable to init input tensor for node: " << *node); - - nvinfer1::ActivationType activate_type; - if (node->kind() == torch::jit::aten::relu || node->kind() == torch::jit::aten::relu_) { - activate_type = nvinfer1::ActivationType::kRELU; - } else if (node->kind() == torch::jit::aten::relu6 || node->kind() == torch::jit::aten::relu6_) { - activate_type = nvinfer1::ActivationType::kRELU; - } else if (node->kind() == torch::jit::aten::sigmoid || node->kind() == torch::jit::aten::sigmoid_) { - activate_type = nvinfer1::ActivationType::kSIGMOID; - } else if (node->kind() == torch::jit::aten::tanh || node->kind() == torch::jit::aten::tanh_) { - activate_type = nvinfer1::ActivationType::kTANH; - } else if (node->kind() == torch::jit::aten::leaky_relu) { - activate_type = nvinfer1::ActivationType::kLEAKY_RELU; - } else if (node->kind() == torch::jit::aten::hardtanh || node->kind() == torch::jit::aten::hardtanh_) { - activate_type = nvinfer1::ActivationType::kCLIP; - } else if (node->kind() == torch::jit::aten::elu) { - activate_type = nvinfer1::ActivationType::kELU; - }else if (node->kind() == torch::jit::aten::silu) { - activate_type = nvinfer1::ActivationType::kSIGMOID; - } else { - POROS_THROW_ERROR("We should never reach here for ActivationConverter, meet Unsupported ActivationType!"); - } - - auto new_layer = engine->network()->addActivation(*nv_tensor, activate_type); - - //set attributes for aten::leaky_relu - //"aten::leaky_relu(Tensor self, Scalar negative_slope=0.01) -> Tensor", - if (activate_type == nvinfer1::ActivationType::kLEAKY_RELU) { - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for aten::leaky_relu in ActivationConverter"); - auto negative_slopeScalar = (engine->context().get_constant(inputs[1])).toScalar().to(); - new_layer->setAlpha(negative_slopeScalar); - } - - //set attributes for aten::hardtanh - //"aten::hardtanh(Tensor self, Scalar min_val=-1, Scalar max_val=1) -> Tensor", - if (activate_type == nvinfer1::ActivationType::kCLIP) { - POROS_CHECK_TRUE((inputs.size() == 3), "invaid inputs size for aten::hardtanh in ActivationConverter"); - auto min = (engine->context().get_constant(inputs[1])).toDouble(); - auto max = (engine->context().get_constant(inputs[2])).toDouble(); - new_layer->setAlpha(min); - new_layer->setBeta(max); - } - - //set attributes for aten::elu - //"aten::elu(Tensor self, Scalar alpha=1, Scalar scale=1, Scalar input_scale=1) -> Tensor" - if (activate_type == nvinfer1::ActivationType::kELU) { - POROS_CHECK_TRUE((inputs.size() == 4), "invaid inputs size for aten::hardtanh in ActivationConverter"); - auto alpha = (engine->context().get_constant(inputs[1])).toDouble(); - new_layer->setAlpha(alpha); - } - - new_layer->setName((layer_info(node) + "_IActivationLayer").c_str()); - nvinfer1::ITensor* output = new_layer->getOutput(0); - if (node->kind() == torch::jit::aten::relu6 || node->kind() == torch::jit::aten::relu6_) { - nvinfer1::ITensor* relu_output = new_layer->getOutput(0); - auto options_pyt = torch::TensorOptions().device(torch::kCUDA, 0).dtype(at::kFloat); - at::Tensor relu6_max = at::tensor({6.0}, options_pyt); - nvinfer1::ITensor* relu6_max_nv = tensor_to_const(engine, relu6_max); - - auto min_layer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kMIN, - relu_output, - relu6_max_nv, - layer_info(node) + "_min"); - output = min_layer->getOutput(0); - }else if (node->kind() == torch::jit::aten::silu) { - nvinfer1::ITensor* sigmoid_output = new_layer->getOutput(0); - auto min_layer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - sigmoid_output, - nv_tensor, - layer_info(node) + "_prod"); - output = min_layer->getOutput(0); - } - - - - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output shape: " << output->getDimensions(); - return true; -} - -// aten::gelu(Tensor self) -> Tensor -// aten::gelu(Tensor self, *, str approximate='none') -> Tensor -bool GeluActivationConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 1 || inputs.size() == 2), "invaid inputs size for GeluActivationConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for GeluActivationConverter is not Tensor as expected"); - - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), - "Unable to init input tensor for node: " << *node); - nvinfer1::DataType type = in->getType(); - - POROS_CHECK((type == nvinfer1::DataType::kFLOAT || type == nvinfer1::DataType::kHALF), - "gelu only supports kFLOAT and kHALF"); - - std::string pluginName = "CustomGeluPluginDynamic"; - nvinfer1::PluginFieldCollection fc; - std::vector f; - - //TODO: maybe need to consider more about op_precision situation - // int type_id = ctx->settings.op_precision == nvinfer1::DataType::kFLOAT - // ? 0 - // : 1; // Integer encoding the DataType (0: FP32, 1: FP16) - int type_id = (type == nvinfer1::DataType::kFLOAT) ? 0 : 1; - f.emplace_back(nvinfer1::PluginField("type_id", &type_id, nvinfer1::PluginFieldType::kINT32, 1)); - - std::string mode = "gelu"; - f.emplace_back(nvinfer1::PluginField("mode", &mode, nvinfer1::PluginFieldType::kCHAR, 1)); - - fc.nbFields = f.size(); - fc.fields = f.data(); - - auto creator = getPluginRegistry()->getPluginCreator("CustomGeluPluginDynamic", "1", ""); - auto gelu_plugin = creator->createPlugin("gelu", &fc); - - POROS_CHECK(gelu_plugin, "Unable to create gelu plugin from TensorRT plugin registry" << *node); - auto new_layer = - engine->network()->addPluginV2(reinterpret_cast(&in), 1, *gelu_plugin); - new_layer->setName((layer_info(node) + "_plugin_gelu").c_str()); - auto out_tensor = new_layer->getOutput(0); - engine->context().set_tensor(node->outputs()[0], out_tensor); - LOG(INFO) << "Output shape: " << out_tensor->getDimensions(); - return true; -} - -/*"aten::prelu(Tensor self, Tensor weight) -> Tensor"*/ -bool PreluActivationConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for PreluActivationConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for PreluActivationConverter is not Tensor as expected"); - - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - - auto maybe_slopes = engine->context().get_constant(inputs[1]); - POROS_CHECK_TRUE((maybe_slopes.isTensor()), "Unable to init input const-tensor for node: " << *node); - auto slopes = maybe_slopes.toTensor(); //at::tensor - //auto slopes_size = sizes_to_nvdim(slopes.sizes()); - - //bool to_reshape = false; - auto original_shape = in->getDimensions(); - - // Channel dim is the 2nd dim of input. When input has dims < 2, then there is no channel dim and the number of channels = 1. - at::Tensor weight; - if (slopes.numel() != 1){ - std::vector weights; - std::vector reshape_shape; - bool sign = true; - for (int i = 0; i < original_shape.nbDims; i++) { - if (original_shape.d[i] == slopes.numel() && sign) { - sign = false; - continue; - } - if (!sign) { - reshape_shape.push_back(original_shape.d[i]); - } - } - - for (int64_t i = 0; i < slopes.numel(); i++) { - auto options_pyt = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kFloat32); - auto tmp = at::ones(reshape_shape, options_pyt); - weights.push_back((slopes[i] * tmp).unsqueeze(0)); - } - - weight = torch::cat(weights, 0); - weight = weight.unsqueeze(0); - } else { - weight = slopes; - } - - auto slope_tensor = tensor_to_const(engine, weight); - auto new_layer = engine->network()->addParametricReLU(*in, *slope_tensor); - new_layer->setName((layer_info(node) + "_IParametricReLULayer").c_str()); - auto out_tensor = new_layer->getOutput(0); - - engine->context().set_tensor(node->outputs()[0], out_tensor); - LOG(INFO) << "Output shape: " << out_tensor->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, ActivationConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, GeluActivationConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, PreluActivationConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/activation.h b/poros/poros/converter/gpu/activation.h deleted file mode 100644 index 8ce274e1086..00000000000 --- a/poros/poros/converter/gpu/activation.h +++ /dev/null @@ -1,115 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file activation.h -* @author tianjinjin@baidu.com -* @date Wed Jul 28 15:24:51 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include -#include - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class ActivationConverter : public GpuConverter { -public: - ActivationConverter() {} - virtual ~ActivationConverter() {} - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::relu(Tensor self) -> Tensor", - "aten::relu_(Tensor(a!) self) -> Tensor(a!)", - "aten::relu6(Tensor self) -> (Tensor)", - "aten::relu6_(Tensor(a!) self) -> Tensor(a!)", - "aten::sigmoid(Tensor self) -> Tensor", - "aten::sigmoid_(Tensor(a!) self) -> Tensor(a!)", - "aten::tanh(Tensor self) -> Tensor", - "aten::tanh_(Tensor(a!) self) -> Tensor(a!)", - "aten::leaky_relu(Tensor self, Scalar negative_slope=0.01) -> Tensor", - "aten::hardtanh(Tensor self, Scalar min_val=-1, Scalar max_val=1) -> Tensor", - "aten::hardtanh_(Tensor(a!) self, Scalar min_val=-1, Scalar max_val=1) -> Tensor(a!)", - "aten::elu(Tensor self, Scalar alpha=1, Scalar scale=1, Scalar input_scale=1) -> Tensor", - "aten::silu(Tensor self) -> Tensor"}; - } - - /** TODO: TRY TO SUPPORT SCHEMA PATTERNS BELLOW: - * //said 'leaky_relu_' is not a member of 'torch::jit::aten' and i don't know why - * "aten::leaky_relu_(Tensor(a!) self, Scalar negative_slope=0.01) -> Tensor(a!)", - * */ - const std::vector node_kind() { - return {torch::jit::aten::relu, - torch::jit::aten::relu_, - torch::jit::aten::relu6, - torch::jit::aten::relu6_, - torch::jit::aten::sigmoid, - torch::jit::aten::sigmoid_, - torch::jit::aten::tanh, - torch::jit::aten::tanh_, - torch::jit::aten::leaky_relu, - torch::jit::aten::hardtanh, - torch::jit::aten::hardtanh_, - torch::jit::aten::elu, - torch::jit::aten::silu}; - } -}; - -class GeluActivationConverter : public GpuConverter { -public: - GeluActivationConverter() {} - virtual ~GeluActivationConverter() {} - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - const std::vector schema_string() { - // aten::gelu schema changed in torch-1.12 - if (TORCH_VERSION_MAJOR < 2 && TORCH_VERSION_MINOR < 12) { - return {"aten::gelu(Tensor self) -> Tensor"}; - } else { - return {"aten::gelu(Tensor self, *, str approximate='none') -> Tensor"}; - } - } - - const std::vector node_kind() { - return {torch::jit::aten::gelu}; - } -}; - -class PreluActivationConverter : public GpuConverter { -public: - PreluActivationConverter() {} - virtual ~PreluActivationConverter() {} - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - const std::vector schema_string() { - return {"aten::prelu(Tensor self, Tensor weight) -> Tensor"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::prelu}; - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/add.cpp b/poros/poros/converter/gpu/add.cpp deleted file mode 100644 index 32e530a4954..00000000000 --- a/poros/poros/converter/gpu/add.cpp +++ /dev/null @@ -1,377 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file add.cpp -* @author tianjinjin@baidu.com -* @date Mon Mar 8 11:36:11 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/add.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/** - * @brief unify type : According to the type promotion rules of torch, - * when one of self or other is float type, the other also becomes float. - * @param [in] engine : trt engine - * @param [in] node : current node - * @param [in] self : self ITensor - * @param [in] other : other ITensor - * @return -**/ -static void unify_type(TensorrtEngine* engine, - const torch::jit::Node *node, - nvinfer1::ITensor*& self, - nvinfer1::ITensor*& other) { - if (self->getType() == nvinfer1::DataType::kFLOAT && - other->getType() == nvinfer1::DataType::kINT32) { - auto id_layer = engine->network()->addIdentity(*other); - id_layer->setOutputType(0, nvinfer1::DataType::kFLOAT); - id_layer->setName((layer_info(node) + "_IIdentityLayer_other_to_float").c_str()); - other = id_layer->getOutput(0); - } - - if (other->getType() == nvinfer1::DataType::kFLOAT && - self->getType() == nvinfer1::DataType::kINT32) { - auto id_layer = engine->network()->addIdentity(*self); - id_layer->setOutputType(0, nvinfer1::DataType::kFLOAT); - id_layer->setName((layer_info(node) + "_IIdentityLayer_self_to_float").c_str()); - self = id_layer->getOutput(0); - } -} - -/* -"aten::add.Tensor(Tensor self, Tensor other, Scalar alpha=1) -> Tensor", -"aten::add.Scalar(Tensor self, Scalar other, Scalar alpha=1) -> Tensor"*/ -bool AddConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - - // aten::add.int(int a, int b) -> (int) - // aten::add.t(t[] a, t[] b) -> (t[]) - if (node->schema().operator_name() == torch::jit::parseSchema(this->schema_string()[4]).operator_name() || - node->schema().operator_name() == torch::jit::parseSchema(this->schema_string()[5]).operator_name()) { - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for AddConverter"); - if (check_inputs_tensor_scalar(engine, node)) { - // 获取int对应的nvtensor - nvinfer1::ITensor* a = this->get_tensor_scalar(inputs[0]); - nvinfer1::ITensor* b = this->get_tensor_scalar(inputs[1]); - // 判断是否为空 (get_constant失败时可能为空) - // 为空时返回false, 让子图fallback - POROS_CHECK_TRUE((a != nullptr && b != nullptr), - node_info(node) + std::string("get nvtensor type int false.")); - if (node->schema().operator_name() == torch::jit::parseSchema(this->schema_string()[4]).operator_name()) { - // a和b相加并返回 - nvinfer1::ILayer* add_layer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - a, b, layer_info(node) + "_sum"); - POROS_CHECK(add_layer, "Unable to create add layer from node: " << *node); - nvinfer1::ITensor* output = add_layer->getOutput(0); - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - } else { - std::vector inputs_nvtensor; - // 将所有int对应的nvtensor加入vector, 最后cat起来 - inputs_nvtensor.push_back(a); - inputs_nvtensor.push_back(b); - nvinfer1::IConcatenationLayer* concat_layer = - engine->network()->addConcatenation(inputs_nvtensor.data(), inputs_nvtensor.size()); - concat_layer->setAxis(0); - concat_layer->setName((layer_info(node) + "_IConcatenationLayer").c_str()); - engine->context().set_tensor(node->outputs()[0], concat_layer->getOutput(0)); - } - } else { - torch::jit::IValue a_ivalue = engine->context().get_constant(inputs[0]); - if (a_ivalue.isInt()) { - int a = engine->context().get_constant(inputs[0]).toScalar().to(); - int b = engine->context().get_constant(inputs[1]).toScalar().to(); - engine->context().set_constant(node->outputs()[0], a + b); - } else if (a_ivalue.isIntList()) { - std::vector a_vec = engine->context().get_constant(inputs[0]).toIntList().vec(); - std::vector b_vec = engine->context().get_constant(inputs[1]).toIntList().vec(); - a_vec.insert(a_vec.end(), b_vec.begin(), b_vec.end()); - auto output_ivalue = c10::optional(std::move(torch::jit::IValue(a_vec))); - engine->context().set_constant(node->outputs()[0], output_ivalue); - } else { - // a and b are tensorlists - if (inputs[0]->type()->isSubtypeOf(c10::ListType::ofTensors())) { - std::vector in_tensor_a, in_tensor_b; - engine->context().get_tensorlist(inputs[0], in_tensor_a); - engine->context().get_tensorlist(inputs[1], in_tensor_b); - in_tensor_a.insert(in_tensor_a.end(), in_tensor_b.begin(), in_tensor_b.end()); - engine->context().set_tensorlist(node->outputs()[0], in_tensor_a); - } else { - LOG(INFO) << *node->maybeSchema() << " meets unkown input type."; - return false; - } - } - } - return true; - } - - POROS_CHECK_TRUE((inputs.size() == 3), "invaid inputs size for AddConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for AddConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[2]->node()->kind() == torch::jit::prim::Constant), - "input[2] for AddConverter is not come from prim::Constant as expected"); - - //extract self - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - - //extract scalar - //TODO: check input scalar is int type situation - auto scalar_ivalue = (engine->context().get_constant(inputs[2])); - // 先转成float去接input[2]输入 - auto scalar = scalar_ivalue.toScalar().to(); - //which one is better? - //auto scalar = ((engine->context().get_constant(inputs[2]))->to(); - //auto scalar = ((engine->context().get_constant(inputs[2])).to()).to(); - - //extract other - auto other = engine->context().get_tensor(inputs[1]); - //situation1: ---------- when other input is Scalar ------------- - if (other == nullptr) { - auto other_const = engine->context().get_constant(inputs[1]); - if (other_const.isScalar()) { - // 先转成float去接input[1]输入 - auto other_scalar = other_const.toScalar().to(); - at::Tensor prod_tensor = torch::tensor({other_scalar * scalar}); - // input[1]和input[2]若本身是int,相乘结果需转成int - if (scalar_ivalue.isInt() && other_const.isInt()) { - prod_tensor = prod_tensor.to(at::ScalarType::Int); - } - other = tensor_to_const(engine, prod_tensor); - } else { - POROS_THROW_ERROR("Unable to get input other value for AddConverter"); - } - //situation2: ---------- when other input is Tensor ------------- - } else { - const double EPSILON = 1e-9; - if (fabs(scalar - 1.0) > EPSILON) { - nvinfer1::ITensor* alphaTensor = nullptr; - // input[1]和input[2]若本身是int,则input[2]需转回int。否则trt中float和int相乘为0。 - if (scalar_ivalue.isInt() && other->getType() == nvinfer1::DataType::kINT32) { - alphaTensor = tensor_to_const(engine, torch::tensor({scalar}).to(at::ScalarType::Int)); - } else { - alphaTensor = tensor_to_const(engine, torch::tensor({scalar})); - } - auto scaleLayer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - other, - alphaTensor, - layer_info(node) + "_prod"); - POROS_CHECK(scaleLayer, "Unable to create alpha*input layer from node: " << *node); - other = scaleLayer->getOutput(0); - } - } - - unify_type(engine, node, self, other); - - auto add = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - self, - other, - layer_info(node) + "_sum"); - POROS_CHECK(add, "Unable to create add layer from node: " << *node); - engine->context().set_tensor(node->outputs()[0], add->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << add->getOutput(0)->getDimensions(); - return true; -} - -/* -"aten::sub.Tensor(Tensor self, Tensor other, *, Scalar alpha=1) -> Tensor", -"aten::sub.Scalar(Tensor self, Scalar other, Scalar alpha=1) -> Tensor",*/ -bool SubConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - // aten::sub.int(int a, int b) -> (int) - if (node->schema().operator_name() == torch::jit::parseSchema(this->schema_string()[4]).operator_name()) { - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for SubConverter"); - if (check_inputs_tensor_scalar(engine, node)) { - // 获取int对应的nvtensor - nvinfer1::ITensor* a = this->get_tensor_scalar(inputs[0]); - nvinfer1::ITensor* b = this->get_tensor_scalar(inputs[1]); - // 判断是否为空 (get_constant失败时可能为空) - // 为空时返回false, 让子图fallback - POROS_CHECK_TRUE((a != nullptr && b != nullptr), - node_info(node) + std::string("get int nvtensor false.")); - // a和b相加并返回 - nvinfer1::ILayer* sub_layer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUB, - a, b, layer_info(node) + "_sub"); - POROS_CHECK(sub_layer, "Unable to create sub layer from node: " << *node); - nvinfer1::ITensor* output = sub_layer->getOutput(0); - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - } else { - int a = engine->context().get_constant(inputs[0]).toScalar().to(); - int b = engine->context().get_constant(inputs[1]).toScalar().to(); - engine->context().set_constant(node->outputs()[0], a - b); - } - return true; - } - - POROS_CHECK_TRUE((inputs.size() == 3), "invaid inputs size for SubConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for SubConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[2]->node()->kind() == torch::jit::prim::Constant), - "input[2] for SubConverter is not come from prim::Constant as expected"); - - //extract self - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - - //extract scalar - //TODO: check input scalar is int type situation - auto scalar_ivalue = (engine->context().get_constant(inputs[2])); - // 先转成float去接input[2]输入 - auto scalar = scalar_ivalue.toScalar().to(); - - //extract other - auto other = engine->context().get_tensor(inputs[1]); - //situation1: ---------- when other input is Scalar ------------- - if (other == nullptr) { - auto other_const = engine->context().get_constant(inputs[1]); - if (other_const.isScalar()) { - // 先转成float去接input[1]输入 - auto other_scalar = other_const.toScalar().to(); - at::Tensor prod_tensor = torch::tensor({other_scalar * scalar}); - // input[1]和input[2]若本身是int,相乘结果需转成int - if (scalar_ivalue.isInt() && other_const.isInt()) { - prod_tensor = prod_tensor.to(at::ScalarType::Int); - } - other = tensor_to_const(engine, prod_tensor); - } else { - POROS_THROW_ERROR("Unable to get input other value for MulConverter"); - } - //situation2: ---------- when other input is Tensor ------------- - } else { - const double EPSILON = 1e-9; - if (fabs(scalar - 1.0) > EPSILON) { - nvinfer1::ITensor* alphaTensor = nullptr; - // input[1]和input[2]若本身是int,则input[2]需转回int。否则trt中float和int相乘为0。 - if (scalar_ivalue.isInt() && other->getType() == nvinfer1::DataType::kINT32) { - alphaTensor = tensor_to_const(engine, torch::tensor({scalar}).to(at::ScalarType::Int)); - } else { - alphaTensor = tensor_to_const(engine, torch::tensor({scalar})); - } - auto scaleLayer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - other, - alphaTensor, - layer_info(node) + "_prod"); - POROS_CHECK(scaleLayer, "Unable to create alpha*input layer from node: " << *node); - other = scaleLayer->getOutput(0); - } - } - - unify_type(engine, node, self, other); - - auto sub = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUB, - self, - other, - layer_info(node) + "_sub"); - POROS_CHECK(sub, "Unable to create sub layer from node: " << *node); - engine->context().set_tensor(node->outputs()[0], sub->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << sub->getOutput(0)->getDimensions(); - return true; -} - -// aten::rsub.Tensor(Tensor self, Tensor other, Scalar alpha=1) -> (Tensor) -// aten::rsub.Scalar(Tensor self, Scalar other, Scalar alpha=1) -> (Tensor) -bool RsubConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 3), "invaid inputs size for SubConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for SubConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[2]->node()->kind() == torch::jit::prim::Constant), - "input[2] for SubConverter is not come from prim::Constant as expected"); - - //extract self - nvinfer1::ITensor* self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - //extract scalar - //TODO: check input scalar is int type situation - auto scalar_ivalue = engine->context().get_constant(inputs[2]); - // 先转成float去接input[2]输入 - auto scalar = scalar_ivalue.toScalar().to(); - - // self * alpha - const double EPSILON = 1e-9; - if (fabs(scalar - 1.0) > EPSILON) { - nvinfer1::ITensor* alpha_tensor = nullptr; - // input[0]和input[2]若本身是int,则input[2]需转回int。否则trt中float和int相乘为0。 - if (scalar_ivalue.isInt() && self->getType() == nvinfer1::DataType::kINT32) { - alpha_tensor = tensor_to_const(engine, torch::tensor({scalar}).to(at::ScalarType::Int)); - } else { - alpha_tensor = tensor_to_const(engine, torch::tensor({scalar})); - } - auto scaleLayer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - self, - alpha_tensor, - layer_info(node) + "_prod"); - POROS_CHECK(scaleLayer, "Unable to create alpha*input layer from node: " << *node); - self = scaleLayer->getOutput(0); - } - - //extract other - auto other = engine->context().get_tensor(inputs[1]); - //situation1: ---------- when other input is Scalar ------------- - if (other == nullptr) { - auto other_const = engine->context().get_constant(inputs[1]); - if (other_const.isScalar()) { - // 先转成float去接input[1]输入 - auto other_scalar = other_const.toScalar().to(); - at::Tensor other_tensor = torch::tensor({other_scalar}); - if (other_const.isInt() && self->getType() == nvinfer1::DataType::kINT32) { - other_tensor = other_tensor.to(at::ScalarType::Int); - } - other = tensor_to_const(engine, other_tensor); - } else { - POROS_THROW_ERROR("Unable to get input other value for MulConverter"); - } - } - - unify_type(engine, node, self, other); - - auto sub = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUB, - other, - self, - layer_info(node) + "_rsub"); - POROS_CHECK(sub, "Unable to create sub layer from node: " << *node); - engine->context().set_tensor(node->outputs()[0], sub->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << sub->getOutput(0)->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, AddConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, SubConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, RsubConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/converter/gpu/add.h b/poros/poros/converter/gpu/add.h deleted file mode 100644 index c3284c60c1b..00000000000 --- a/poros/poros/converter/gpu/add.h +++ /dev/null @@ -1,120 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file add.h -* @author tianjinjin@baidu.com -* @date Mon Aug 16 12:26:28 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class AddConverter : public GpuConverter { -public: - AddConverter() {} - virtual ~AddConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - //aten::add.Tensor(Tensor self, Tensor other, *, Scalar alpha=1) -> Tensor - const std::vector schema_string() { - return {"aten::add.Tensor(Tensor self, Tensor other, Scalar alpha=1) -> Tensor", - "aten::add.Scalar(Tensor self, Scalar other, Scalar alpha=1) -> Tensor", - "aten::add_.Tensor(Tensor(a!) self, Tensor other, *, Scalar alpha=1) -> Tensor(a!)", - "aten::add_.Scalar(Tensor(a!) self, Scalar other, Scalar alpha=1) -> Tensor(a!)", - "aten::add.int(int a, int b) -> (int)", - "aten::add.t(t[] a, t[] b) -> (t[])" - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::add.out(Tensor self, Tensor other, *, Scalar alpha=1, Tensor(a!) out) -> Tensor(a!)", - * **/ - const std::vector node_kind() { - return {torch::jit::aten::add, - torch::jit::aten::add_}; - } - - bool assign_schema_attr() { - bool result = true; - result &= assign_schema_attr_helper({{"aten::add.int(int a, int b) -> (int)", {1, 1}}}); - result &= assign_schema_attr_helper({{"aten::add.t(t[] a, t[] b) -> (t[])", {1, 1}}}); - result &= assign_schema_attr_helper({{"aten::add.Scalar(Tensor self, Scalar other, Scalar alpha=1) -> (Tensor)", {1, 1}}}); - return result; - } -}; - -class SubConverter : public GpuConverter { -public: - SubConverter() {} - virtual ~SubConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::sub.Tensor(Tensor self, Tensor other, *, Scalar alpha=1) -> Tensor", - "aten::sub.Scalar(Tensor self, Scalar other, Scalar alpha=1) -> Tensor", - "aten::sub_.Tensor(Tensor(a!) self, Tensor other, *, Scalar alpha=1) -> Tensor(a!)", - "aten::sub_.Scalar(Tensor(a!) self, Scalar other, Scalar alpha=1) -> Tensor(a!)", - "aten::sub.int(int a, int b) -> (int)", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::sub.out(Tensor self, Tensor other, *, Scalar alpha=1, Tensor(a!) out) -> Tensor(a!)", - * **/ - const std::vector node_kind() { - return {torch::jit::aten::sub, - torch::jit::aten::sub_}; - } - - bool assign_schema_attr() { - return assign_schema_attr_helper({{"aten::sub.int(int a, int b) -> (int)", {1, 1}}}); - } -}; - -class RsubConverter : public GpuConverter { -public: - RsubConverter() {} - virtual ~RsubConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::rsub.Tensor(Tensor self, Tensor other, Scalar alpha=1) -> (Tensor)", - "aten::rsub.Scalar(Tensor self, Scalar other, Scalar alpha=1) -> (Tensor)", - }; - } - - const std::vector node_kind() { - return {torch::jit::aten::rsub}; - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/aten_eval.cpp b/poros/poros/converter/gpu/aten_eval.cpp deleted file mode 100644 index c61a1640085..00000000000 --- a/poros/poros/converter/gpu/aten_eval.cpp +++ /dev/null @@ -1,164 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file aten_eval.cpp -* @author tianjinjin@baidu.com -* @date Mon Mar 8 11:36:11 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/aten_eval.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/* -"aten::append.t(t[](a!) self, t(c -> *) el) -> (t[](a!))"*/ -bool AppendConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for AppendConverter"); - - //extract self - std::vector tensorlist; - POROS_CHECK_TRUE((engine->context().get_tensorlist(inputs[0], tensorlist)), "extract tensor list err"); - // 防止输入的tensorlist没有经过trtengine的情况 - //(不加的话 build trtengine 时会报 Unused Input 或者 Tensor xxx cannot be both input and output 的错误) - for (size_t i = 0; i < tensorlist.size(); i++) { - tensorlist[i] = engine->network()->addIdentity(*tensorlist[i])->getOutput(0); - } - //extract element - auto element = engine->context().get_tensor(inputs[1]); - //element is an already changed nvtensor - if (element != nullptr) { - element = engine->network()->addIdentity(*element)->getOutput(0); - tensorlist.emplace_back(element); - engine->context().set_tensorlist(node->outputs()[0], tensorlist); - engine->context().set_tensorlist(node->inputs()[0], tensorlist); - return true; - } else { - LOG(WARNING) << "non tensor kind element append is currently not support in AppendConverter"; - return false; - } -} - -/* -"aten::__getitem__.t(t[](a) list, int idx) -> (t(*))"*/ -bool GetitemConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for GetitemConverter"); - POROS_CHECK_TRUE((inputs[1]->node()->kind() == torch::jit::prim::Constant), - "inputs[1] for GetitemConverter is not come from prim::Constant as expected"); - - if (node->outputs()[0]->type()->str() == "Tensor") { - //extract list - std::vector tensorlist; - POROS_CHECK_TRUE((engine->context().get_tensorlist(inputs[0], tensorlist)), "extract tensor list err") - - const int64_t list_size = tensorlist.size(); - auto index = (engine->context().get_constant(inputs[1])).toInt(); - const int64_t normalized_index = index < 0 ? list_size + index : index; - nvinfer1::ITensor* out_tensor = tensorlist[normalized_index]; - engine->context().set_tensor(node->outputs()[0], out_tensor); - return true; - } else { - // extract nvtensor intlist - if (check_inputs_tensor_scalar(engine, node)) { - nvinfer1::ITensor* list_itensor = this->get_tensor_scalar(inputs[0]); - POROS_CHECK_TRUE((list_itensor != nullptr), - node_info(node) + std::string("get int nvtensor false.")); - - auto index = (engine->context().get_constant(inputs[1])).toInt(); - auto list_len = (list_itensor->getDimensions()).d[0]; - POROS_CHECK_TRUE((index >= -list_len && index <= list_len - 1), - node_info(node) + std::string(" idx is out of range.")); - - // 倒序改正序 - index = index < 0 ? index + list_len : index; - - nvinfer1::ITensor* index_itensor = tensor_to_const(engine, torch::tensor({index}, torch::kInt)); - - //extract the specific dynamic dim as a 1D-1value tensor - std::vector start_vec{0}, size_vec{1}, stride_vec{1}; - nvinfer1::ISliceLayer* slice_layer = engine->network()->addSlice(*list_itensor, - sizes_to_nvdim(start_vec), - sizes_to_nvdim(size_vec), - sizes_to_nvdim(stride_vec)); - POROS_CHECK(slice_layer, "Unable to given dim info from node: " << *node); - slice_layer->setInput(1, *index_itensor); - slice_layer->setName((layer_info(node) + "_ISliceLayer").c_str()); - nvinfer1::ITensor* slice_out = slice_layer->getOutput(0); - engine->context().set_tensor(node->outputs()[0], slice_out); - LOG(INFO) << "Output tensor shape: " << slice_out->getDimensions(); - } else { - torch::jit::IValue ts_ivalue = engine->context().get_constant(inputs[0]); - POROS_CHECK_TRUE((ts_ivalue.isList()), "Unable to init input tensor for node: " << *node); - auto list = ts_ivalue.toListRef(); - const int64_t list_size = list.size(); - int64_t index = (engine->context().get_constant(inputs[1])).toInt(); - const int64_t normalized_index = index < 0 ? list_size + index : index; - auto value_item = list[normalized_index]; - engine->context().set_constant(node->outputs()[0], value_item); - } - return true; - } -} - -/* -"aten::_set_item.t(t[](a!) l, int idx, t(b -> *) el) -> (t[](a!))"*/ -bool SetitemConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 3), "invaid inputs size for SetitemConverter"); - POROS_CHECK_TRUE((inputs[1]->node()->kind() == torch::jit::prim::Constant), - "inputs[1] for SetitemConverter is not come from prim::Constant as expected"); - - size_t idx = engine->context().get_constant(inputs[1]).toInt(); - - if (node->outputs()[0]->type()->str() == "Tensor[]") { - std::vector tensorlist; - POROS_CHECK_TRUE((engine->context().get_tensorlist(inputs[0], tensorlist)), "extract tensor list err"); - POROS_CHECK_TRUE((tensorlist.size() > idx), "Tensorlist index out of range: " << *node); - // 防止输入的tensorlist没有经过trtengine的情况 - //(不加的话 build trtengine 时会报 Unused Input 或者 Tensor xxx cannot be both input and output 的错误) - for (size_t i = 0; i < tensorlist.size(); i++) { - tensorlist[i] = engine->network()->addIdentity(*tensorlist[i])->getOutput(0); - } - nvinfer1::ITensor *input_tensor = engine->context().get_tensor(inputs[2]); - input_tensor = engine->network()->addIdentity(*input_tensor)->getOutput(0); - tensorlist[idx] = input_tensor; - engine->context().set_tensorlist(node->outputs()[0], tensorlist); - engine->context().set_tensorlist(node->inputs()[0], tensorlist); - return true; - } - else{ - LOG(WARNING) << "non tensor kind element _set_item is currently not support in SetitemConverter"; - return true; - } -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, AppendConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, GetitemConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, SetitemConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/aten_eval.h b/poros/poros/converter/gpu/aten_eval.h deleted file mode 100644 index fbc34c47915..00000000000 --- a/poros/poros/converter/gpu/aten_eval.h +++ /dev/null @@ -1,93 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file aten_eval.h -* @author tianjinjin@baidu.com -* @date Mon Aug 16 12:26:28 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class AppendConverter : public GpuConverter { -public: - AppendConverter() {} - virtual ~AppendConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - //aten::add.Tensor(Tensor self, Tensor other, *, Scalar alpha=1) -> Tensor - const std::vector schema_string() { - return {"aten::append.t(t[](a!) self, t(c -> *) el) -> (t[](a!))" }; - } - - const std::vector node_kind() { - return {torch::jit::aten::append}; - } -}; - -class GetitemConverter : public GpuConverter { -public: - GetitemConverter() {} - virtual ~GetitemConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - //aten::__getitem__.t(t[](a) list, int idx) -> (t(*)) - const std::vector schema_string() { - return {"aten::__getitem__.t(t[](a) list, int idx) -> (t(*))"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::__getitem__}; - } - - bool assign_schema_attr() { - return assign_schema_attr_helper({{"aten::__getitem__.t(t[](a) list, int idx) -> (t(*))", {1, 1}}}); - } -}; - -class SetitemConverter : public GpuConverter { -public: - SetitemConverter() {} - virtual ~SetitemConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - //aten::_set_item.t(t[](a!) l, int idx, t(b -> *) el) -> (t[](a!)) - const std::vector schema_string() { - return {"aten::_set_item.t(t[](a!) l, int idx, t(b -> *) el) -> (t[](a!))"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::_set_item}; - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/aten_trt_util.cpp b/poros/poros/converter/gpu/aten_trt_util.cpp deleted file mode 100644 index aaea55019a7..00000000000 --- a/poros/poros/converter/gpu/aten_trt_util.cpp +++ /dev/null @@ -1,87 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file aten_trt_util.cpp -* @author tianjinjin@baidu.com -* @date Fri Aug 6 14:17:11 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/aten_trt_util.h" -#include "poros/engine/trtengine_util.h" -#include "poros/util/macros.h" - -namespace baidu { -namespace mirana { -namespace poros { - -//将torchscript中的at::tensor转变成tensorrt中的weights结构 -bool at_tensor_to_trt_weignts(at::Tensor tensor, nvinfer1::Weights& weight) { - POROS_CHECK_TRUE((tensor.sizes().size() <= nvinfer1::Dims::MAX_DIMS), - "given tensor is outof max_dims"); - - /* - auto shape = sizes_to_nvdim(tensor.sizes()); - //TODO: CHECK this bias info. - int64_t inputs_num = (tensor.sizes().size() > 1) ? tensor.sizes()[1] : tensor.sizes()[0]; - int64_t outputs_num = tensor.sizes()[0]; - - nvinfer1::Dims kernel_shape; - if (tensor.sizes().size() > 2) { - kernel_shape.nbDims = tensor.sizes().size() - 2; - for (size_t i = 2; i < tensor.sizes().size(); i++) { - kernel_shape.d[i - 2] = tensor.size()[i]; - } - } else { - kernal_shape.nbdims = 1; - kernal_shape.d[0] = 1; - }*/ - - auto t_cpu = tensor.to(at::kCPU); - t_cpu = t_cpu.contiguous(); - - auto t_type = c10::optTypeMetaToScalarType(t_cpu.dtype()); - POROS_CHECK_TRUE(t_type.has_value(), "unsupported datatype"); - //TODO: may be failed here - auto dtype = attype_to_nvtype(t_type.value()); - - void* buf = nullptr; - if (dtype == nvinfer1::DataType::kFLOAT) { - buf = malloc(t_cpu.numel() * sizeof(float)); - memcpy(buf, t_cpu.data_ptr(), t_cpu.numel() * sizeof(float)); - } else if (dtype == nvinfer1::DataType::kHALF) { - buf = malloc(t_cpu.numel() * (sizeof(float) / 2)); - memcpy(buf, t_cpu.data_ptr(), t_cpu.numel() * (sizeof(float) / 2)); - } else if (dtype == nvinfer1::DataType::kINT8) { - buf = malloc(t_cpu.numel() * sizeof(char)); - memcpy(buf, t_cpu.data_ptr(), t_cpu.numel() * sizeof(char)); - } else if (dtype == nvinfer1::DataType::kINT32) { - buf = malloc(t_cpu.numel() * sizeof(int)); - memcpy(buf, t_cpu.data_ptr(), t_cpu.numel() * sizeof(int)); - } else if (dtype == nvinfer1::DataType::kBOOL) { - buf = malloc(t_cpu.numel() * sizeof(bool)); - memcpy(buf, t_cpu.data_ptr(), t_cpu.numel() * sizeof(bool)); - } - - weight.type = dtype; - weight.count = t_cpu.numel(); - weight.values = buf; - - return true; -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/aten_trt_util.h b/poros/poros/converter/gpu/aten_trt_util.h deleted file mode 100644 index 34d4ddb7e76..00000000000 --- a/poros/poros/converter/gpu/aten_trt_util.h +++ /dev/null @@ -1,38 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file aten_trt_util.h -* @author tianjinjin@baidu.com -* @date Fri Aug 6 10:42:39 CST 2021 -* @brief -**/ - -#pragma once - -#include - -#include "torch/script.h" -#include "NvInfer.h" - -namespace baidu { -namespace mirana { -namespace poros { - -//将torchscript中的at::tensor转变成tensorrt中的weights结构 -bool at_tensor_to_trt_weignts(at::Tensor tensor, nvinfer1::Weights& weight); - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/converter/gpu/batch_norm.cpp b/poros/poros/converter/gpu/batch_norm.cpp deleted file mode 100644 index b39e11dac27..00000000000 --- a/poros/poros/converter/gpu/batch_norm.cpp +++ /dev/null @@ -1,248 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file batch_norm.cpp -* @author tianjinjin@baidu.com -* @date Sun Aug 15 22:23:03 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/batch_norm.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/converter/gpu/weight.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/* -aten::batch_norm(Tensor input, -Tensor? weight, -Tensor? bias, -Tensor? running_mean, -Tensor? running_var, -bool training, -float momentum, -float eps, -bool cudnn_enabled) -> Tensor -*/ -bool BatchNormConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - - POROS_CHECK_TRUE((inputs.size() == 9), "invaid inputs size for BatchNormConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for BatchNormConverter is not Tensor as expected"); - // weight & bias & running_mean & running_var - POROS_CHECK_TRUE((inputs[1]->node()->kind() == torch::jit::prim::Constant), - "input[1] for BatchNormConverter is not come from prim::Constant as expected"); - POROS_CHECK_TRUE((inputs[2]->node()->kind() == torch::jit::prim::Constant), - "input[2] for BatchNormConverter is not come from prim::Constant as expected"); - POROS_CHECK_TRUE((inputs[3]->node()->kind() == torch::jit::prim::Constant), - "input[3] for BatchNormConverter is not come from prim::Constant as expected"); - POROS_CHECK_TRUE((inputs[4]->node()->kind() == torch::jit::prim::Constant), - "input[4] for BatchNormConverter is not come from prim::Constant as expected"); - - auto input = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((input != nullptr), - "Unable to init input tensor for node: " << *node); - auto orig_shape = input->getDimensions(); - auto shape = nvdim_to_sizes(orig_shape); - auto tensor_type = nvtype_to_attype(input->getType()); - auto options = torch::TensorOptions().dtype(tensor_type); - - torch::Tensor gamma, beta, mean, var; - auto maybe_gamma = engine->context().get_constant(inputs[1]); - auto maybe_beta = engine->context().get_constant(inputs[2]); - auto maybe_mean = engine->context().get_constant(inputs[3]); - auto maybe_bar = engine->context().get_constant(inputs[4]); - - if (maybe_gamma.isTensor()) { - gamma = maybe_gamma.toTensor(); - } else { - gamma = at::full({shape}, 1, {options}); - } - - if (maybe_beta.isTensor()) { - beta = maybe_beta.toTensor(); - } else { - beta = at::full({shape}, 1, {options}); - } - - if (maybe_mean.isTensor()) { - mean = maybe_mean.toTensor(); - } else { - mean = at::full({shape}, 0, {options}); - } - - if (maybe_bar.isTensor()) { - var = maybe_bar.toTensor(); - } else { - var = at::full({shape}, 0, {options}); - } - - auto eps = engine->context().get_constant(inputs[7]).to(); - - // Expand spatial dims from 1D to 2D if needed - bool expandDims = (orig_shape.nbDims < 4); - if (expandDims) { - input = add_padding(engine, node, input, 4); - } - - auto scale = gamma / torch::sqrt(var + eps); - auto bias = beta - mean * scale; - - auto scale_weights = Weights(scale); - auto bias_weights = Weights(bias); - - auto power = Weights(at::ones_like(scale)); - auto bn = engine->network()->addScaleNd( - *input, nvinfer1::ScaleMode::kCHANNEL, bias_weights.data, scale_weights.data, power.data, 1); - bn->setName((layer_info(node) + "_IScaleLayer").c_str()); - // Un-pad bn output if needed - auto out_tensor = add_unpadding(engine, node, bn->getOutput(0), orig_shape.nbDims); - engine->context().set_tensor(node->outputs()[0], out_tensor); - LOG(INFO) << "Output tensor shape: " << out_tensor->getDimensions(); - return true; -} - -/* -aten::instance_norm(Tensor input, -Tensor? weight, -Tensor? bias, -Tensor? running_mean, -Tensor? running_var, -bool use_input_stats, -float momentum, -float eps, -bool cudnn_enabled) -> Tensor -*/ -bool InstanceNormConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - - POROS_CHECK_TRUE((inputs.size() == 9), "invaid inputs size for InstanceNormConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for InstanceNormConverter is not Tensor as expected"); - // weight & bias & running_mean & running_var - POROS_CHECK_TRUE((inputs[1]->node()->kind() == torch::jit::prim::Constant), - "input[1] for InstanceNormConverter is not come from prim::Constant as expected"); - POROS_CHECK_TRUE((inputs[2]->node()->kind() == torch::jit::prim::Constant), - "input[2] for InstanceNormConverter is not come from prim::Constant as expected"); - POROS_CHECK_TRUE((inputs[3]->node()->kind() == torch::jit::prim::Constant), - "input[3] for InstanceNormConverter is not come from prim::Constant as expected"); - POROS_CHECK_TRUE((inputs[4]->node()->kind() == torch::jit::prim::Constant), - "input[4] for InstanceNormConverter is not come from prim::Constant as expected"); - - auto input = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((input != nullptr), - "Unable to init input tensor for node: " << *node); - auto orig_shape = input->getDimensions(); - auto shape = nvdim_to_sizes(orig_shape); - auto tensor_type = nvtype_to_attype(input->getType()); - auto options = torch::TensorOptions().dtype(tensor_type); - - // Expand spatial dims from 1D to 2D if needed - bool expand_dims = (orig_shape.nbDims < 4); - if (expand_dims) { - input = add_padding(engine, node, input, 4); - } - - torch::Tensor weight, bias, mean, var; - auto maybe_weight = engine->context().get_constant(inputs[1]); - auto maybe_bias = engine->context().get_constant(inputs[2]); - auto maybe_mean = engine->context().get_constant(inputs[3]); - auto maybe_var = engine->context().get_constant(inputs[4]); - - if (maybe_weight.isTensor()) { - weight = maybe_weight.toTensor().cpu().contiguous(); - } else { - weight = at::ones(shape[1], options).cpu().contiguous(); - } - - if (maybe_bias.isTensor()) { - bias = maybe_bias.toTensor().cpu().contiguous(); - } else { - bias = at::zeros(shape[1], options).cpu().contiguous(); - } - - auto eps = static_cast(engine->context().get_constant(inputs[7]).toDouble()); - - //TODO: 确认此处设置 ”或“ 还是 “与” 合适 - if (maybe_mean.isTensor() && maybe_var.isTensor()) { - mean = maybe_mean.toTensor(); - var = maybe_var.toTensor(); - - auto scale = weight.to(mean.options()) / torch::sqrt(var + eps); - auto new_bias = bias.to(mean.options()) - mean * scale; - - auto scale_weights = Weights(scale); - auto bias_weights = Weights(new_bias); - - auto power = Weights(at::ones_like(scale)); - auto bn = engine->network()->addScaleNd( - *input, nvinfer1::ScaleMode::kCHANNEL, bias_weights.data, scale_weights.data, power.data, 1); - bn->setName((layer_info(node) + "_IScaleLayer").c_str()); - // Un-pad bn output if needed - auto out_tensor = add_unpadding(engine, node, bn->getOutput(0), orig_shape.nbDims); - engine->context().set_tensor(node->outputs()[0], out_tensor); - LOG(INFO) << "Output tensor shape: " << out_tensor->getDimensions(); - return true; - } - - // https://github.com/NVIDIA/TensorRT/tree/release/8.4/plugin/instanceNormalizationPlugin - - const int relu = 0; - const float alpha = 0; - std::vector f; - f.emplace_back(nvinfer1::PluginField("epsilon", &eps, nvinfer1::PluginFieldType::kFLOAT32, 1)); - f.emplace_back(nvinfer1::PluginField( - "scales", weight.data_ptr(), nvinfer1::PluginFieldType::kFLOAT32, weight.numel())); - f.emplace_back(nvinfer1::PluginField( - "bias", bias.data_ptr(), nvinfer1::PluginFieldType::kFLOAT32, bias.numel())); - f.emplace_back(nvinfer1::PluginField("relu", &relu, nvinfer1::PluginFieldType::kINT32, 1)); - f.emplace_back(nvinfer1::PluginField("alpha", &alpha, nvinfer1::PluginFieldType::kFLOAT32, 1)); - - nvinfer1::PluginFieldCollection fc; - fc.nbFields = f.size(); - fc.fields = f.data(); - - auto creator = getPluginRegistry()->getPluginCreator("InstanceNormalization_TRT", "1", ""); - auto instance_norm_plugin = creator->createPlugin("instance_norm", &fc); - - POROS_CHECK(instance_norm_plugin, "Unable to create instance_norm plugin from TensorRT plugin registry" << *node); - auto new_layer = engine->network()->addPluginV2( - reinterpret_cast(&input), 1, *instance_norm_plugin); - new_layer->setName((layer_info(node) + "_plugin_instance_norm").c_str()); - nvinfer1::ITensor* output = new_layer->getOutput(0); - - if (expand_dims) { - output = add_unpadding(engine, node, output, orig_shape.nbDims); - } - - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, BatchNormConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, InstanceNormConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/batch_norm.h b/poros/poros/converter/gpu/batch_norm.h deleted file mode 100644 index dba8bfeea48..00000000000 --- a/poros/poros/converter/gpu/batch_norm.h +++ /dev/null @@ -1,71 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file batch_norm.h -* @author tianjinjin@baidu.com -* @date Fri Aug 13 16:51:50 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class BatchNormConverter : public GpuConverter { -public: - BatchNormConverter() {} - virtual ~BatchNormConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - virtual const std::vector schema_string() { - return {"aten::batch_norm(Tensor input, Tensor? weight, Tensor? bias, Tensor? running_mean, Tensor? running_var, bool training, float momentum, float eps, bool cudnn_enabled) -> Tensor"}; - } - - virtual const std::vector node_kind() { - return {torch::jit::aten::batch_norm,}; - } -}; - - -class InstanceNormConverter : public GpuConverter { -public: - InstanceNormConverter() {} - virtual ~InstanceNormConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - virtual const std::vector schema_string() { - return {"aten::instance_norm(Tensor input, Tensor? weight, Tensor? bias, Tensor? running_mean, Tensor? running_var, bool use_input_stats, float momentum, float eps, bool cudnn_enabled) -> Tensor"}; - } - - virtual const std::vector node_kind() { - return {torch::jit::aten::instance_norm,}; - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/clone.cpp b/poros/poros/converter/gpu/clone.cpp deleted file mode 100644 index 9ebcf3a52a6..00000000000 --- a/poros/poros/converter/gpu/clone.cpp +++ /dev/null @@ -1,92 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file clone.cpp -* @author tianshaoqing@baidu.com -* @date Tue Nov 23 12:26:28 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/clone.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -// aten::clone(Tensor self, *, MemoryFormat? memory_format=None) -> Tensor -bool CloneConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for CloneConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for CloneConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[1]->node()->kind() == torch::jit::prim::Constant), - "input[1] for CloneConverter is not come from prim::Constant as expected"); - - POROS_CHECK_TRUE(engine->context().get_constant(inputs[1]).isNone(), - "not support memory format set yet."); - - //extract self - nvinfer1::ITensor* self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - - // select whole input tensor to clone a new tensor - nvinfer1::Dims self_dims = self->getDimensions(); - bool is_dynamic = check_nvtensor_is_dynamic(self); - - std::vector start_vec, size_vec, stride_vec; - for (int32_t i = 0; i < self_dims.nbDims; i++) { - start_vec.push_back(0); - if (is_dynamic) { - size_vec.push_back(0); - } else { - size_vec.push_back(self_dims.d[i]); - } - stride_vec.push_back(1); - } - - nvinfer1::Dims start_dim = sizes_to_nvdim(start_vec); - nvinfer1::Dims size_dim = sizes_to_nvdim(size_vec); - nvinfer1::Dims stride_dim = sizes_to_nvdim(stride_vec); - - nvinfer1::ITensor* self_shape = nullptr; - if (is_dynamic) { - self_shape = engine->network()->addShape(*self)->getOutput(0); - } - - nvinfer1::ISliceLayer* slice_layer = engine->network()->addSlice(*self, start_dim, size_dim, stride_dim); - POROS_CHECK(slice_layer, "Unable to create slice layer from node: " << *node); - if (is_dynamic) { - slice_layer->setInput(2, *self_shape); - } - slice_layer->setName((layer_info(node) + "_ISliceLayer").c_str()); - nvinfer1::ITensor* output = slice_layer->getOutput(0); - - engine->context().set_tensor(node->outputs()[0], self); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, CloneConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/converter/gpu/clone.h b/poros/poros/converter/gpu/clone.h deleted file mode 100644 index 92f6617c43f..00000000000 --- a/poros/poros/converter/gpu/clone.h +++ /dev/null @@ -1,55 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file clone.h -* @author tianshaoqing@baidu.com -* @date Tue Nov 23 12:26:28 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class CloneConverter : public GpuConverter { -public: - CloneConverter() {} - virtual ~CloneConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - // aten::clone(Tensor self, *, MemoryFormat? memory_format=None) -> Tensor - const std::vector schema_string() { - return {"aten::clone(Tensor self, *, MemoryFormat? memory_format=None) -> Tensor"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::clone}; - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/coercion.cpp b/poros/poros/converter/gpu/coercion.cpp deleted file mode 100644 index afe442d9437..00000000000 --- a/poros/poros/converter/gpu/coercion.cpp +++ /dev/null @@ -1,60 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file coercion.cpp -* @author wangrui39@baidu.com -* @date Fri May 13 11:36:11 CST 2022 -* @brief -**/ - -#include "poros/converter/gpu/coercion.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/*"aten::Int.float(float a) -> (int)" -"aten::Int.Tensor(Tensor a) -> (int)*/ -bool CoercionConverter::converter(TensorrtEngine *engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 1), "invaid inputs size for CoercionConverter"); - nvinfer1::ITensor *tensor_a = engine->context().get_tensor(inputs[0]); - - // int to tensor - if (nullptr != tensor_a) { - auto id_layer = engine->network()->addIdentity(*tensor_a); - id_layer->setName((layer_info(node) + "_IIdentityLayer").c_str()); - id_layer->setOutputType(0, nvinfer1::DataType::kINT32); - engine->context().set_tensor(node->outputs()[0], id_layer->getOutput(0)); - } else { - int a = engine->context().get_constant(inputs[0]).toScalar().to(); - engine->context().set_constant(node->outputs()[0], a); - } - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, CoercionConverter); - - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/coercion.h b/poros/poros/converter/gpu/coercion.h deleted file mode 100644 index 5eaddf45f88..00000000000 --- a/poros/poros/converter/gpu/coercion.h +++ /dev/null @@ -1,63 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file coercion.h -* @author wangrui39@baidu.com -* @date Fri May 13 11:36:11 CST 2022 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class CoercionConverter : public GpuConverter { -public: - CoercionConverter() {} - virtual ~CoercionConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::Int.float(float a) -> (int)", - "aten::Int.Tensor(Tensor a) -> (int)"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::Int}; - } - - bool assign_schema_attr() { - bool result = true; - result &= assign_schema_attr_helper({{"aten::Int.float(float a) -> (int)", {1, 1}}}); - result &= assign_schema_attr_helper({{"aten::Int.Tensor(Tensor a) -> (int)", {1, 1}}}); - return result; - } - -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/concat.cpp b/poros/poros/converter/gpu/concat.cpp deleted file mode 100644 index f25b7236e38..00000000000 --- a/poros/poros/converter/gpu/concat.cpp +++ /dev/null @@ -1,65 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file concat.cpp -* @author tianjinjin@baidu.com -* @date Mon Mar 8 11:36:11 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/concat.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/converter/gpu/weight.h" -#include "poros/context/poros_global.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/*"aten::cat(Tensor[] tensors, int dim=0) -> Tensor*/ -bool ConcatConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for ConcatConverter"); - POROS_CHECK_TRUE(inputs[0]->type()->isSubtypeOf(c10::ListType::ofTensors()), - "input[0] for ConcatConverter is not TensorList as expected"); - POROS_CHECK_TRUE(inputs[1]->type()->isSubtypeOf(c10::NumberType::get()), - "input[1] for ConcatConverter is not int64_t as expected"); - - std::vector tensorlist; - POROS_CHECK_TRUE((engine->context().get_tensorlist(inputs[0], tensorlist)), "extract tensor list err") - - //extract dims - auto dim = (engine->context().get_constant(inputs[1])).toInt(); - if (dim < 0) { - dim = tensorlist[0]->getDimensions().nbDims + dim; - } - - auto cat_layer = engine->network()->addConcatenation(tensorlist.data(), tensorlist.size()); - cat_layer->setAxis(static_cast(dim)); - cat_layer->setName((layer_info(node) + "_IConcatenationLayer").c_str()); - engine->context().set_tensor(node->outputs()[0], cat_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << cat_layer->getOutput(0)->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, ConcatConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/concat.h b/poros/poros/converter/gpu/concat.h deleted file mode 100644 index 147d7603ab9..00000000000 --- a/poros/poros/converter/gpu/concat.h +++ /dev/null @@ -1,60 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file concat.h -* @author tianjinjin@baidu.com -* @date Tue Jul 27 11:24:21 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -//TODO: there is a concat_opt.cpp in torchscript. check it. -class ConcatConverter : public GpuConverter { -public: - ConcatConverter() {} - virtual ~ConcatConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::cat(Tensor[] tensors, int dim=0) -> Tensor"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::cat}; - } - - bool assign_schema_attr() { - return assign_schema_attr_helper({{"aten::cat(Tensor[] tensors, int dim=0) -> Tensor", {1, 1}}}); - } - -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/constant.cpp b/poros/poros/converter/gpu/constant.cpp deleted file mode 100644 index f0735adfdd3..00000000000 --- a/poros/poros/converter/gpu/constant.cpp +++ /dev/null @@ -1,80 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file constant.cpp -* @author tianjinjin@baidu.com -* @date Mon Mar 8 11:36:11 CST 2021 -* @brief -**/ -#include "torch/script.h" - -#include "poros/converter/gpu/constant.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/converter/gpu/weight.h" -#include "poros/context/poros_global.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -bool ConstantConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - c10::optional ivalue = toIValue(node->output()); - POROS_CHECK_TRUE(ivalue.has_value(), "invaid data for ConstantConverter"); - engine->context().set_constant(node->outputs()[0], ivalue.value()); - - //situation1: Tensor - if (ivalue.value().isTensor()) { - auto tensor = ivalue.value().toTensor(); - auto t_weights = Weights(tensor); - auto const_layer = engine->network()->addConstant(t_weights.shape, t_weights.data); - const_layer->setName(layer_info(node).c_str()); - engine->context().set_tensor(node->outputs()[0], const_layer->getOutput(0)); - } - //situation2: Tensor[] - else if(ivalue.value().isTensorList()) { - auto c10_tensorlist = ivalue.value().toTensorList(); - std::vector tensorlist; - tensorlist.reserve(c10_tensorlist.size()); - for (size_t i = 0; i < c10_tensorlist.size(); i++){ - nvinfer1::ITensor* nv_tensor = tensor_to_const(engine, c10_tensorlist[i]); - tensorlist.emplace_back(nv_tensor); - } - engine->context().set_tensorlist(node->outputs()[0], tensorlist); - } - //situation3: Tensor?[] - else if (ivalue.value().type()->str().find("Tensor?[]") != std::string::npos) { - c10::List c10_tensorlist = ivalue.value().to>(); - std::vector tensorlist; - tensorlist.reserve(c10_tensorlist.size()); - for (size_t i = 0; i < c10_tensorlist.size(); i++){ - auto tensor = c10_tensorlist.get(i).toTensor(); - nvinfer1::ITensor* nv_tensor = tensor_to_const(engine, tensor); - tensorlist.emplace_back(nv_tensor); - } - engine->context().set_tensorlist(node->outputs()[0], tensorlist); - } - - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, ConstantConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/constant.h b/poros/poros/converter/gpu/constant.h deleted file mode 100644 index d17218cdbbe..00000000000 --- a/poros/poros/converter/gpu/constant.h +++ /dev/null @@ -1,60 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file constant.h -* @author tianjinjin@baidu.com -* @date Tue Jul 27 11:24:21 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class ConstantConverter : public GpuConverter { -public: - ConstantConverter() {} - virtual ~ConstantConverter() {} - - /** - * @brief converter, 将node转换成layer,添加到engine的组网逻辑里 - * @retval 0 => success, <0 => fail - **/ - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - //prim::Constant kind node has no schema - const std::vector schema_string() { - return {}; - } - - const std::vector node_kind() { - return {torch::jit::prim::Constant}; - } - -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/constant_pad_nd.cpp b/poros/poros/converter/gpu/constant_pad_nd.cpp deleted file mode 100644 index f1f5527acd8..00000000000 --- a/poros/poros/converter/gpu/constant_pad_nd.cpp +++ /dev/null @@ -1,375 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file constant_pad_nd.cpp -* @author tianshaoqing@baidu.com -* @date Thur Dec 2 14:29:20 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/constant_pad_nd.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -//DEPRECATED: 该实现方式内部采用的contat,在trt的profile阶段,会额外引入一些copy节点,导致性能变差。 -// aten::constant_pad_nd(Tensor self, int[] pad, Scalar value=0) -> Tensor -bool ConstantPadNdConverter::converter_old_version(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 3), "invaid inputs size for ReplicationPadConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for ConstantPadNdConverter is not Tensor as expected"); - - // extract self - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - auto self_dims = self->getDimensions(); - int64_t self_rank = self_dims.nbDims; - - // check input is dynamic or not - std::vector self_dims_vec = nvdim_to_sizes(self_dims); - // extract self - torch::jit::IValue maybe_pad = engine->context().get_constant(inputs[1]); - POROS_CHECK_TRUE((!maybe_pad.isNone()), "invaid inputs[1] for ConstantPadNdConverter"); - std::vector padding = maybe_pad.toIntList().vec(); - int64_t pad_size = padding.size(); - - // pad_size must be an integer multiple of 2 - POROS_CHECK_TRUE((pad_size % 2 == 0), "Length of pad must be even but instead it equals: " << pad_size); - int64_t l_pad = pad_size / 2; - POROS_CHECK_TRUE((self_rank >= l_pad), "Length of pad should be no more than twice the number of " - "dimensions of the input. Pad length is " << pad_size << "while the input has " << self_rank << "dimensions."); - - // extract value - torch::jit::IValue maybe_value = engine->context().get_constant(inputs[2]); - POROS_CHECK_TRUE((!maybe_value.isNone()), "invaid inputs[2] for ConstantPadNdConverter"); - float value = maybe_value.toScalar().to(); - - // prepare for dynamic - const bool is_dynamic = check_nvtensor_is_dynamic(self); - - // dynamic下trt的Ifilllayer无法构建bool类型,所以先返回false(虽然constant_pad_nd bool的非常少) - if (is_dynamic && maybe_value.isBool()) { - LOG(WARNING) << "ConstantPadNdConverter is not support padding value is type of bool when dynamic."; - return false; - } - - nvinfer1::ITensor* self_shape = nullptr; - nvinfer1::ITensor* rev_mask_shape_tensor = nullptr; - if (is_dynamic) { - self_shape = engine->network()->addShape(*self)->getOutput(0); - } - - // create itensors vector - std::vector itensors_vec; - - for (int64_t i = 0; i < l_pad; i++) { - int64_t axis = self_rank - (i + 1); - int64_t padding_index = i * 2; - // dynamic情况,需要使用mask完成padding shape的构造 - // 首先,使用rev_mask_tensor使self_shape[axis] = 0 - // 例如:self_shape = [2, 3, 4, 5],axis = 3,则rev_mask_shape_tensor = [2, 3, 4, 0] - if (is_dynamic && (padding[padding_index] > 0 || padding[padding_index + 1] > 0)) { - at::Tensor rev_mask_tensor = at::ones({self_rank}, torch::kInt); - rev_mask_tensor[axis] = 0; - nvinfer1::ITensor* nv_rev_mask_tensor = tensor_to_const(engine, rev_mask_tensor); - rev_mask_shape_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - self_shape, - nv_rev_mask_tensor, - layer_info(node) + "_prod(axis_dim_to_zero)_" + std::to_string(i))->getOutput(0); - } - - if (padding[padding_index] > 0) { - itensors_vec.clear(); - // 非dynamic情况 - if (!is_dynamic) { - // create pad tensor - self_dims_vec[axis] = padding[padding_index]; - at::Tensor pad_tenosr = at::full(self_dims_vec, value, torch::kFloat32); - // 默认是float32类型,如果self是int32的需转换类型 - if (self->getType() == nvinfer1::DataType::kINT32) { - pad_tenosr = pad_tenosr.to(at::ScalarType::Int); - } - // 默认是float32类型,如果self是bool的需转换类型(bool情况很少) - if (self->getType() == nvinfer1::DataType::kBOOL && maybe_value.isBool()) { - pad_tenosr = pad_tenosr.to(at::ScalarType::Bool); - } - itensors_vec.push_back(tensor_to_const(engine, pad_tenosr)); - } else { - // dynamic情况 - // 然后,使用mask_tensor构造只有axis下标是padding[padding_index],其余数据都是0的tensor - // 例如:self_shape = [2, 3, 4, 5],则self_rank = 4, - // 若当前axis = 3,padding[padding_index] = 2,则构造出来的nv_mask_tensor = [0, 0, 0, 2] - at::Tensor mask_tensor = at::zeros({self_rank}, torch::kInt); - mask_tensor[axis] = padding[padding_index]; - nvinfer1::ITensor* nv_mask_tensor = tensor_to_const(engine, mask_tensor); - // 最后,nv_mask_tensor与之前得到的rev_mask_shape_tensor相加,就得到padding shape - // 例如:刚才rev_mask_shape_tensor = [2, 3, 4, 0], nv_mask_tensor = [0, 0, 0, 2] - // 则pad_shape_tensor = [2, 3, 4, 2] - nvinfer1::ITensor* pad_shape_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - rev_mask_shape_tensor, - nv_mask_tensor, - layer_info(node) + "_sum(gen_left_pad_shape)_" + std::to_string(i))->getOutput(0); - // 根据padding shape和value创建nvtensor - auto fill_layer = engine->network()->addFill(nvinfer1::Dims{1, {1}}, nvinfer1::FillOperation::kLINSPACE); - fill_layer->setInput(0, *pad_shape_tensor); - fill_layer->setName((layer_info(node) + "_IFillLayer_" + std::to_string(padding_index)).c_str()); - - at::Tensor value_tensor = torch::tensor(value, torch::kFloat32); - at::Tensor delta_tensor = torch::zeros(self_rank, torch::kFloat32); - // 默认是float32类型,如果self是int32的需转换类型 - if (self->getType() == nvinfer1::DataType::kINT32) { - value_tensor = value_tensor.to(at::ScalarType::Int); - delta_tensor = delta_tensor.to(at::ScalarType::Int); - } - auto value_itensor = tensor_to_const(engine, value_tensor); - fill_layer->setInput(1, *value_itensor); // 初始值 - auto delta_itensor = tensor_to_const(engine, delta_tensor); - fill_layer->setInput(2, *delta_itensor); // delta值 - - itensors_vec.push_back(fill_layer->getOutput(0)); - } - - itensors_vec.push_back(self); - // concat - nvinfer1::IConcatenationLayer* concat_layer = - engine->network()->addConcatenation(itensors_vec.data(), itensors_vec.size()); - concat_layer->setAxis(axis); - concat_layer->setName((layer_info(node) + "_IConcatenationLayer_" + std::to_string(padding_index)).c_str()); - self = concat_layer->getOutput(0); - // 非dynamic更新维度信息 - self_dims = self->getDimensions(); - self_dims_vec = nvdim_to_sizes(self_dims); - // dynamic更新维度信息 - if (is_dynamic) { - self_shape = engine->network()->addShape(*self)->getOutput(0); - } - } - - if (padding[padding_index + 1] > 0) { - itensors_vec.clear(); - // padding self dim=axis的另一边, - // 与上面的代码只有self加入itensors_vec先后顺序的区别,这里是先push_back - itensors_vec.push_back(self); - - // create pad tensor - if (!is_dynamic) { - self_dims_vec[axis] = padding[padding_index + 1]; - at::Tensor pad_tenosr = at::full(self_dims_vec, value, torch::kFloat32); - // 默认是float32类型,如果self是int32的需转换类型 - if (self->getType() == nvinfer1::DataType::kINT32) { - pad_tenosr = pad_tenosr.to(at::ScalarType::Int); - } - // 默认是float32类型,如果self是bool的需转换类型(bool情况很少) - if (self->getType() == nvinfer1::DataType::kBOOL && maybe_value.isBool()) { - pad_tenosr = pad_tenosr.to(at::ScalarType::Bool); - } - itensors_vec.push_back(tensor_to_const(engine, pad_tenosr)); - } else { - // 与上面代码类似 - at::Tensor mask_tensor = at::zeros({self_rank}, torch::kInt); - mask_tensor[axis] = padding[padding_index + 1]; - nvinfer1::ITensor* nv_mask_tensor = tensor_to_const(engine, mask_tensor); - nvinfer1::ITensor* pad_shape_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - rev_mask_shape_tensor, - nv_mask_tensor, - layer_info(node) + "_sum(gen_right_pad_shape)_" + std::to_string(i))->getOutput(0); - - auto fill_layer = engine->network()->addFill(nvinfer1::Dims{1, {1}}, nvinfer1::FillOperation::kLINSPACE); - fill_layer->setInput(0, *pad_shape_tensor); // 设置output shape - fill_layer->setName((layer_info(node) + "_IFillLayer_more_" + std::to_string(padding_index)).c_str()); - at::Tensor value_tensor = torch::tensor(value, torch::kFloat32); - at::Tensor delta_tensor = torch::zeros(self_rank, torch::kFloat32); // 只有1个维度 - // 默认是float32类型,如果self是int32的需转换类型 - if (self->getType() == nvinfer1::DataType::kINT32) { - value_tensor = value_tensor.to(at::ScalarType::Int); - delta_tensor = delta_tensor.to(at::ScalarType::Int); - } - auto value_itensor = tensor_to_const(engine, value_tensor); - fill_layer->setInput(1, *value_itensor); // 初始值 - auto delta_itensor = tensor_to_const(engine, delta_tensor); - fill_layer->setInput(2, *delta_itensor); - - itensors_vec.push_back(fill_layer->getOutput(0)); - } - - // concat - nvinfer1::IConcatenationLayer* concat_layer = - engine->network()->addConcatenation(itensors_vec.data(), itensors_vec.size()); - concat_layer->setAxis(axis); - concat_layer->setName((layer_info(node) + "_IConcatenationLayer_" + std::to_string(padding_index + 1)).c_str()); - self = concat_layer->getOutput(0); - // 非dynamic更新维度信息 - self_dims = self->getDimensions(); - self_dims_vec = nvdim_to_sizes(self_dims); - // dynamic更新维度信息 - if (is_dynamic) { - self_shape = engine->network()->addShape(*self)->getOutput(0); - } - } - } - - engine->context().set_tensor(node->outputs()[0], self); - LOG(INFO) << "Output tensor shape: " << self->getDimensions(); - - return true; -} - -/** - * @brief 将pytorch组织的padding信息,转化成tensorrt可以接受的padding。 - * pytorch 下的padding order: - * The order is dim_n_begin, dim_n_end, dim_n-1_begin, dim_n-1_end, ..., dim_m_begin, dim_m_end, - * where m is in range [0, n]. - * 期望被转变成的padding order: - * dim_0_begin, dim_1_begin, ... , dim_0_end, ..., dim_n_end. - * while n is the dimension of input. - * 当前的转化逻辑,基于padding 本身是constant 这个前提(被padding的tensor是否dynamic不影响)。 - * 目前不支持padding本身是动态的,如果遇到相应的场景,再添加。 - * padding本身dynamic下的转化思路是: - * padding后面先补0,再reshape成(-1,2)维度,然后flip+transpose,最后reshape成1维即可。 - * torch::reshape(torch::transpose(aten::flip(torch::reshape(padding_tensor, {-1, 2}), [0]), 1, 0),{-1}); - * **/ -bool ConstantPadNdConverter::converter_padding(TensorrtEngine* engine, - int64_t rank, - const std::vector& padding, - nvinfer1::ITensor*& start_tensor, - nvinfer1::ITensor*& total_padding_tensor) { - - std::vector start; - std::vector total_padding; - if (padding.size() % 2U != 0) { - LOG(WARNING) << "padding size should be even but instead it equals: " << padding.size(); - return false; - } - - const int64_t pad_dim_len = static_cast(padding.size() / 2U); - const int64_t diff = rank - pad_dim_len; - if (diff < 0) { - LOG(WARNING) << "padding size should be no more than twice the number of dimensions of the input" - << " , but given padding size is: " << padding.size() - << " , given input dimensions is: " << rank << "."; - return false; - } - - start.resize(rank, 0); - total_padding.resize(rank, 0); - - for (int64_t i = diff; i < rank; i++) { - const int64_t idx = i - diff; - const int64_t reverse_idx = pad_dim_len - idx - 1; - const int64_t before = padding[reverse_idx * 2]; - const int64_t after = padding[reverse_idx * 2 + 1]; - if (before < 0 || after < 0) { - return false; - } - start[i] = -before; - total_padding[i] = before + after; - } - - at::Tensor at_start_tensor = torch::from_blob(start.data(), start.size(), - torch::TensorOptions().dtype(torch::kInt64)); - start_tensor = tensor_to_const(engine, at_start_tensor); - - at::Tensor at_total_padding_tensor = torch::from_blob(total_padding.data(), total_padding.size(), - torch::TensorOptions().dtype(torch::kInt64)); - total_padding_tensor = tensor_to_const(engine, at_total_padding_tensor); - return start_tensor && total_padding_tensor; -} - -// aten::constant_pad_nd(Tensor self, int[] pad, Scalar value=0) -> Tensor -bool ConstantPadNdConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 3), "invaid inputs size for ReplicationPadConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for ConstantPadNdConverter is not Tensor as expected"); - - // extract self - nvinfer1::ITensor* self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - nvinfer1::Dims self_dims = self->getDimensions(); - const int64_t self_rank = self_dims.nbDims; - - // extract pad - torch::jit::IValue maybe_pad = engine->context().get_constant(inputs[1]); - POROS_CHECK_TRUE((!maybe_pad.isNone()), "invaid inputs[1] for ConstantPadNdConverter"); - std::vector padding = maybe_pad.toIntList().vec(); - - // extract value - torch::jit::IValue maybe_value = engine->context().get_constant(inputs[2]); - POROS_CHECK_TRUE((!maybe_value.isNone()), "invaid inputs[2] for ConstantPadNdConverter"); - nvinfer1::ITensor* value_tensor = nullptr; - float value = maybe_value.toScalar().to(); - // value的类型与self对齐 - if (self->getType() == nvinfer1::DataType::kINT32) { - value_tensor = tensor_to_const(engine, torch::tensor({value}).to(at::ScalarType::Int)); - } else if (self->getType() == nvinfer1::DataType::kBOOL && maybe_value.isBool()) { - value_tensor = tensor_to_const(engine, torch::tensor({value}).to(at::ScalarType::Bool)); - } else { - value_tensor = tensor_to_const(engine, torch::tensor({value}).to(at::ScalarType::Float)); - } - - // const bool is_dynamic = check_nvtensor_is_dynamic(self); - // // dynamic下trt的Ifilllayer无法构建bool类型,所以先返回false(虽然constant_pad_nd bool的非常少) - // if (is_dynamic && maybe_value.isBool()) { - // LOG(WARNING) << "ConstantPadNdConverter is not support padding value is type of bool when dynamic."; - // return false; - // } - - nvinfer1::ITensor* start = nullptr; - nvinfer1::ITensor* total_padding = nullptr; - if (converter_padding(engine, self_rank, padding, start, total_padding) == false) { - return false; - } - nvinfer1::ITensor* self_shape = engine->network()->addShape(*self)->getOutput(0); - nvinfer1::ITensor* size = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - self_shape, - total_padding, - layer_info(node) + "_sum(for_padding)")->getOutput(0); - - //fix stride setting - nvinfer1::Dims stride; - stride.nbDims = self_rank; - std::fill_n(stride.d, self_rank, 1); - const nvinfer1::Dims& dummy = stride; - nvinfer1::ISliceLayer* layer = engine->network()->addSlice(*self, dummy, dummy, stride); - layer->setInput(1, *start); - layer->setInput(2, *size); - layer->setMode(nvinfer1::SliceMode::kFILL); - layer->setInput(4, *value_tensor); - layer->setName((layer_info(node) + "_ISliceLayer").c_str()); - - engine->context().set_tensor(node->outputs()[0], layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << layer->getOutput(0)->getDimensions(); - - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, ConstantPadNdConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/converter/gpu/constant_pad_nd.h b/poros/poros/converter/gpu/constant_pad_nd.h deleted file mode 100644 index b56176ef98e..00000000000 --- a/poros/poros/converter/gpu/constant_pad_nd.h +++ /dev/null @@ -1,78 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file constant_pad_nd.h -* @author tianshaoqing@baidu.com -* @date Thur Dec 2 14:29:20 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class ConstantPadNdConverter : public GpuConverter { -public: - ConstantPadNdConverter() {} - virtual ~ConstantPadNdConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - //DEPRECATED: 该实现方式内部采用的contat,在trt的profile阶段,会额外引入一些copy节点,导致性能变差。 - bool converter_old_version(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::constant_pad_nd(Tensor self, int[] pad, Scalar value=0) -> Tensor"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::constant_pad_nd}; - } - - bool assign_schema_attr() { - return assign_schema_attr_helper({{"aten::constant_pad_nd(Tensor self, int[] pad, Scalar value=0) -> Tensor", {1, 0}}}); - } - -private: - /** - * @brief 将pytorch组织的padding信息,转化成tensorrt可以接受的padding。 - * @param [in] engine : 略 - * @param [in] rank : 被padding的tensor的rank信息(也就是nbDims值) - * @param [in] padding : pytorch序的padding信息,是 int[] 类型(注意: pytorch 的padding信息是从后往前的) - * @param [out] start_tensor : 用于slice 的start tensor信息 - * @param [out] total_padding_tensor : padding 引入后每一维增多的size信息。 - * @return bool - * @retval true => succeed false => failed - * **/ - bool converter_padding(TensorrtEngine* engine, - int64_t rank, - const std::vector& padding, - nvinfer1::ITensor*& start_tensor, - nvinfer1::ITensor*& total_padding_tensor); -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/converter_util.cpp b/poros/poros/converter/gpu/converter_util.cpp deleted file mode 100644 index 6c60849da1b..00000000000 --- a/poros/poros/converter/gpu/converter_util.cpp +++ /dev/null @@ -1,376 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -// Part of the following code in this file refs to -// https://github.com/pytorch/TensorRT/blob/master/core/conversion/converters/converter_util.cpp -// -// Copyright (c) 2020-present, NVIDIA CORPORATION. All rights reserved. -// Copyright (c) Meta Platforms, Inc. and affiliates. -// Licensed under the 3-Clause BSD License - -/** -* @file converter_util.cpp -* @author tianjinjin@baidu.com -* @date Thu Aug 12 14:50:37 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/converter_util.h" -#include "poros/converter/gpu/weight.h" -#include "poros/engine/trtengine_util.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - - -nvinfer1::ITensor* add_padding(TensorrtEngine* engine, - const torch::jit::Node* n, - nvinfer1::ITensor* tensor, - int nDim, - bool trailing, - bool use_zeros) { - const auto dims = tensor->getDimensions(); - if (dims.nbDims < nDim) { - auto newDims = dims; - for (int dim = dims.nbDims; dim < nDim; ++dim) { - newDims = unsqueeze_dims(newDims, trailing ? dim : 0, 1, use_zeros); - } - LOG(INFO) << "Original shape: " << dims << ", reshaping to: " << newDims; - auto shuffle_layer = engine->network()->addShuffle(*tensor); - POROS_CHECK(shuffle_layer, "Unable to create shuffle layer"); - shuffle_layer->setReshapeDimensions(newDims); - shuffle_layer->setZeroIsPlaceholder(use_zeros); - shuffle_layer->setName((layer_info(n) + " [Reshape to " + nvdim_to_str(newDims) + ']').c_str()); - return shuffle_layer->getOutput(0); - } else { - return tensor; - } -} - -nvinfer1::ITensor* add_unpadding(TensorrtEngine* engine, - const torch::jit::Node* n, - nvinfer1::ITensor* tensor, - int nDim, - bool trailing, - bool use_zeros) { - const auto dims = tensor->getDimensions(); - if (dims.nbDims > nDim) { - auto newDims = dims; - for (int dim = dims.nbDims; dim > nDim; --dim) { - newDims = squeeze_dims(newDims, trailing ? dim - 1 : 0); - } - LOG(INFO) << "Original shape: " << dims << ", reshaping to: " << newDims; - auto shuffle_layer = engine->network()->addShuffle(*tensor); - POROS_CHECK(shuffle_layer, "Unable to create shuffle layer"); - shuffle_layer->setReshapeDimensions(newDims); - shuffle_layer->setZeroIsPlaceholder(use_zeros); - shuffle_layer->setName((layer_info(n) + " [Reshape to " + nvdim_to_str(newDims) + "]").c_str()); - return shuffle_layer->getOutput(0); - } else { - return tensor; - } -} - -bool check_tensor_type(nvinfer1::ITensor* &self, nvinfer1::ITensor* &other) { - // 保证二元操作的两个tensor类型相同 float32 > harf > int32 > int8 - if (self->getType() != nvinfer1::DataType::kBOOL){ - if (self->getType() < other->getType()) { - if (self->getType() == nvinfer1::DataType::kINT8){ - self->setType(other->getType()); - } - else { - other->setType(self->getType()); - } - } - else if(other->getType() < self->getType()) { - if (other->getType() == nvinfer1::DataType::kINT8){ - other->setType(self->getType()); - } - else { - self->setType(other->getType()); - } - } - } - return true; -} - -nvinfer1::ILayer* add_elementwise(TensorrtEngine* engine, - nvinfer1::ElementWiseOperation op, - nvinfer1::ITensor* self, - nvinfer1::ITensor* other, - const std::string& name) { - // ensure self to have larger number of dimension - bool swapSelfOther = false; - check_tensor_type(self, other); - if (self->getDimensions().nbDims < other->getDimensions().nbDims) { - std::swap(self, other); - swapSelfOther = true; - } - auto selfDim = nvdim_to_sizes(self->getDimensions()); - auto otherDim = nvdim_to_sizes(other->getDimensions()); - if (selfDim.size() != otherDim.size()) { - // other is with dynamic shape, need to expand its dimension now and get its - // shape at runtime - // 对other而言,如果其dim是-1,则需要保持原维度,如果其dim是1,则需要等于self相应的维度。 - if (otherDim.end() != std::find(otherDim.begin(), otherDim.end(), -1)) { - auto thOtherStaticShapeMask = torch::ones(selfDim.size(), torch::kInt32); - auto thOtherDynamicShapeMask = torch::zeros(selfDim.size(), torch::kInt32); - for (size_t start = selfDim.size() - otherDim.size(), idx = 0; idx < otherDim.size(); ++idx) { - if (-1 != otherDim[idx]) { - thOtherStaticShapeMask[start + idx] = otherDim[idx]; - } else { - thOtherStaticShapeMask[start + idx] = 0; - if (selfDim[start + idx] == 1) { - thOtherDynamicShapeMask[start + idx] = -1; - } else { - thOtherDynamicShapeMask[start + idx] = 1; - } - } - } - auto otherStaticShapeMask = tensor_to_const(engine, thOtherStaticShapeMask); - auto otherDynamicShapeMask = tensor_to_const(engine, thOtherDynamicShapeMask); - auto selfShape = engine->network()->addShape(*self)->getOutput(0); - - // size of dynamic dimension of other need to the same as that of - // corresponding dimension of self - auto otherDynamicShape = engine->network()->addElementWise(*selfShape, - *otherDynamicShapeMask, nvinfer1::ElementWiseOperation::kPROD)->getOutput(0); - auto targetOtherShape = engine->network()->addElementWise(*otherDynamicShape, - *otherStaticShapeMask, nvinfer1::ElementWiseOperation::kSUM)->getOutput(0); - - auto otherShuffle = engine->network()->addShuffle(*other); - otherShuffle->setName((name + "_IShuffleLayer").c_str()); - otherShuffle->setInput(1, *targetOtherShape); - other = otherShuffle->getOutput(0); - } else { - // other is with static shape, expand dimension to make tow tensor have - // the same number of dimension - auto otherShuffle = engine->network()->addShuffle(*other); - otherShuffle->setReshapeDimensions(sizes_to_nvdim_with_pad(otherDim, selfDim.size())); - otherShuffle->setName((name + "_IShuffleLayer").c_str()); - other = otherShuffle->getOutput(0); - } - } - if (swapSelfOther) { - // swap back - std::swap(self, other); - swapSelfOther = false; - } - auto ele = engine->network()->addElementWise(*self, *other, op); - ele->setName(name.c_str()); - return ele; -} - -nvinfer1::ITensor* broadcast_itensor(TensorrtEngine* engine, - const torch::jit::Node* n, - nvinfer1::ITensor* tensor, - const int new_rank, - std::string name) { - int current_rank = tensor->getDimensions().nbDims; - POROS_CHECK((current_rank <= new_rank), "Cannot broadcast a higher rank tensor to a lower rank tensor."); - if (current_rank < new_rank) { - //1. get shape tensor - nvinfer1::ITensor* shape_tensor = engine->network()->addShape(*tensor)->getOutput(0); - - //2.padding the missing rank part with value 1. - std::vector padding_vec(new_rank - current_rank, 1); - nvinfer1::Dims padding_dim = sizes_to_nvdim(c10::IntArrayRef(padding_vec)); - at::Tensor the_padding = torch::tensor(nvdim_to_sizes(padding_dim), torch::kInt32); - nvinfer1::ITensor* padding_shape = tensor_to_const(engine, the_padding); - - //3. concat the shape tensor - std::vector to_concat_tensors = {padding_shape, shape_tensor}; - nvinfer1::IConcatenationLayer* shape_cat_layer = engine->network()->addConcatenation(to_concat_tensors.data(), to_concat_tensors.size()); - shape_cat_layer->setName((layer_info(n) + "_IConcatenationLayer_for_" + name).c_str()); - auto new_shape = shape_cat_layer->getOutput(0); - - //4. shuffle given tensor to the new shape - nvinfer1::IShuffleLayer* reshape_layer = engine->network()->addShuffle(*tensor); - reshape_layer->setInput(1, *new_shape); - reshape_layer->setName((layer_info(n) + "_IShuffleLayer_for_" + name).c_str()); - nvinfer1::ITensor* new_tensor = reshape_layer->getOutput(0); - return new_tensor; - } - return tensor; -} - -nvinfer1::ITensor* cast_itensor(TensorrtEngine* engine, - nvinfer1::ITensor* tensor, - nvinfer1::DataType dtype) { - if (tensor->getType() != dtype) { - std::ostringstream tensor_id; - tensor_id << reinterpret_cast(tensor); - - auto id_layer = engine->network()->addIdentity(*tensor); - POROS_CHECK(id_layer, "Unable to create identity layer for ITensor: " << tensor_id.str()); - auto casted_tensor = id_layer->getOutput(0); - casted_tensor->setType(dtype); - - LOG(INFO) << "Casting ITensor " << tensor_id.str() << " from " << tensor->getType() << " to " << dtype; - std::stringstream ss; - ss << "[Cast ITensor " << tensor_id.str() << " from " << tensor->getType() << " to " << dtype << "]"; - id_layer->setName(ss.str().c_str()); - return casted_tensor; - } else { - return tensor; - } -} - -// 对nv shape tensor进行unsqueeze操作, 支持dim倒序 -nvinfer1::ITensor* unsqueeze_nv_shapetensor(TensorrtEngine* engine, - nvinfer1::ITensor* input, int dim) { - nvinfer1::Dims input_dims = input->getDimensions(); - - if (input_dims.nbDims != 1 || input->getType() != nvinfer1::DataType::kINT32) { - LOG(INFO) << "input is not shape tensor"; - return nullptr; - } - // dim must be in range of [-input_dims.d[0] - 1, input_dims.d[0]]. - if (dim < -input_dims.d[0] - 1 || dim > input_dims.d[0]) { - LOG(INFO) << "expected to be in range of [" << -input_dims.d[0] - 1 << "," - << input_dims.d[0] << "], but got " << dim; - return nullptr; - } - if (dim < 0) { - dim = input_dims.d[0] + dim + 1; - } - std::vector inputs_nvtensor; - nvinfer1::ITensor* insert_tensor = tensor_to_const(engine, torch::tensor({1}, torch::kInt)); - // if dim == 0 or dim == input_dims.d[0], concat origin tensor and insert tensor directly. - if (dim == 0) { - inputs_nvtensor.push_back(insert_tensor); - inputs_nvtensor.push_back(input); - - } else if (dim == input_dims.d[0]) { - inputs_nvtensor.push_back(input); - inputs_nvtensor.push_back(insert_tensor); - } else { - // divide origin tensor into two parts, then insert the unsqueeze tensor. - std::vector start_vec{0}, size_vec{dim}, stride_vec{1}; - nvinfer1::ISliceLayer* slice_front = engine->network()->addSlice(*input, - sizes_to_nvdim(start_vec), - sizes_to_nvdim(size_vec), - sizes_to_nvdim(stride_vec)); - inputs_nvtensor.push_back(slice_front->getOutput(0)); - inputs_nvtensor.push_back(insert_tensor); - start_vec[0] = dim; - size_vec[0] = input_dims.d[0] - dim; - nvinfer1::ISliceLayer* slice_back = engine->network()->addSlice(*input, - sizes_to_nvdim(start_vec), - sizes_to_nvdim(size_vec), - sizes_to_nvdim(stride_vec)); - inputs_nvtensor.push_back(slice_back->getOutput(0)); - } - nvinfer1::IConcatenationLayer* concat_layer = - engine->network()->addConcatenation(inputs_nvtensor.data(), inputs_nvtensor.size()); - concat_layer->setAxis(0); - return concat_layer->getOutput(0); -} - -// 对nv shape tensor进行squeeze操作, 支持dim倒序 -// note: 使用前须检查 input[dim] == 1 -nvinfer1::ITensor* squeeze_nv_shapetensor(TensorrtEngine* engine, - nvinfer1::ITensor* input, int dim) { - nvinfer1::Dims input_dims = input->getDimensions(); - - if (input_dims.nbDims != 1 || input->getType() != nvinfer1::DataType::kINT32) { - LOG(INFO) << "input is not shape tensor"; - return nullptr; - } - // dim must be in range of [-input_dims.d[0], input_dims.d[0] - 1]. - if (dim < -input_dims.d[0] || dim > input_dims.d[0] - 1) { - LOG(INFO) << "expected to be in range of [" << -input_dims.d[0] << "," - << input_dims.d[0] - 1 << "], but got " << dim; - return nullptr; - } - if (dim < 0) { - dim = input_dims.d[0] + dim; - } - std::vector inputs_nvtensor; - //nvinfer1::ITensor* insert_tensor = tensor_to_const(engine, torch::tensor({1}, torch::kInt)); - tensor_to_const(engine, torch::tensor({1}, torch::kInt)); - // if dim == 0 or dim == input_dims.d[0] - 1, slice squeeze dimension directly. - std::vector start_vec{0}, size_vec{input_dims.d[0] - 1}, stride_vec{1}; - if (dim == 0 || dim == input_dims.d[0] - 1) { - if (dim == 0) { - start_vec[0] = 1; - } - nvinfer1::ISliceLayer* slice_l = engine->network()->addSlice(*input, - sizes_to_nvdim(start_vec), - sizes_to_nvdim(size_vec), - sizes_to_nvdim(stride_vec)); - return slice_l->getOutput(0); - - } else { - // divide origin tensor into two parts (skip the squeeze dim), and concat them. - std::vector start_vec{0}, size_vec{dim}, stride_vec{1}; - nvinfer1::ISliceLayer* slice_front = engine->network()->addSlice(*input, - sizes_to_nvdim(start_vec), - sizes_to_nvdim(size_vec), - sizes_to_nvdim(stride_vec)); - inputs_nvtensor.push_back(slice_front->getOutput(0)); - start_vec[0] = dim + 1; - size_vec[0] = input_dims.d[0] - dim - 1; - nvinfer1::ISliceLayer* slice_back = engine->network()->addSlice(*input, - sizes_to_nvdim(start_vec), - sizes_to_nvdim(size_vec), - sizes_to_nvdim(stride_vec)); - inputs_nvtensor.push_back(slice_back->getOutput(0)); - } - nvinfer1::IConcatenationLayer* concat_layer = - engine->network()->addConcatenation(inputs_nvtensor.data(), inputs_nvtensor.size()); - concat_layer->setAxis(0); - return concat_layer->getOutput(0); -} - -nvinfer1::ITensor* unsqueeze_itensor(TensorrtEngine* engine, - nvinfer1::ITensor* input, - const std::vector& axes) { - nvinfer1::ITensor* input_shape_tensor = engine->network()->addShape(*input)->getOutput(0); - int input_rank = input->getDimensions().nbDims; - - const std::set axes_set(axes.begin(), axes.end()); - if (input_rank + axes_set.size() > nvinfer1::Dims::MAX_DIMS) - { - return nullptr; - } - - // compute interlacing subscripts. - std::vector subscripts(input_rank); - std::iota(subscripts.begin(), subscripts.end(), 0); - for (const auto& axis : axes_set) - { - subscripts.insert(subscripts.begin() + axis, input_rank); - } - at::Tensor indices = torch::tensor(subscripts, torch::kInt32); - auto indices_tensor = tensor_to_const(engine, indices); - - //calculate gather(concat(input_shape_tensor, {1}), indices_tensor) - torch::Tensor the_one = torch::tensor(std::vector({1}), torch::kInt32); - nvinfer1::ITensor* one_tensor = tensor_to_const(engine, the_one); - nvinfer1::ITensor* const args[2] = {input_shape_tensor, one_tensor}; - nvinfer1::ITensor* tmp_concat_tensor = engine->network()->addConcatenation(args, 2)->getOutput(0); - nvinfer1::ITensor* new_shape_tensor = engine->network()->addGather(*tmp_concat_tensor, *indices_tensor, 0)->getOutput(0); - - nvinfer1::IShuffleLayer* reshape_layer = engine->network()->addShuffle(*input); - reshape_layer->setInput(1, *new_shape_tensor); - return reshape_layer->getOutput(0); -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/converter_util.h b/poros/poros/converter/gpu/converter_util.h deleted file mode 100644 index 66492eda3b5..00000000000 --- a/poros/poros/converter/gpu/converter_util.h +++ /dev/null @@ -1,87 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file converter_util.h -* @author tianjinjin@baidu.com -* @date Thu Aug 12 10:50:28 CST 2021 -* @brief -**/ - -#pragma once - -#include - -#include "torch/script.h" -#include "NvInfer.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - - -nvinfer1::ITensor* add_padding(TensorrtEngine* engine, - const torch::jit::Node* n, - nvinfer1::ITensor* tensor, - int nDim, - bool trailing = true, - bool use_zeros = true); - -nvinfer1::ITensor* add_unpadding(TensorrtEngine* engine, - const torch::jit::Node* n, - nvinfer1::ITensor* tensor, - int nDim, - bool trailing = true, - bool use_zeros = true); - -nvinfer1::ILayer* add_elementwise(TensorrtEngine* engine, - nvinfer1::ElementWiseOperation op, - nvinfer1::ITensor* self, - nvinfer1::ITensor* other, - const std::string& name); - -nvinfer1::ITensor* broadcast_itensor(TensorrtEngine* engine, - const torch::jit::Node* n, - nvinfer1::ITensor* tensor, - const int new_rank, - std::string name); - -//If an ITensor is of a type not dtype, add an Identity layer to cast it to dtype -nvinfer1::ITensor* cast_itensor(TensorrtEngine* engine, - nvinfer1::ITensor* tensor, - nvinfer1::DataType dtype); - -// 对nv shape tensor进行unsqueeze操作, 支持dim倒序 -nvinfer1::ITensor* unsqueeze_nv_shapetensor(TensorrtEngine* engine, - nvinfer1::ITensor* input, - int dim); -// 对nv shape tensor进行squeeze操作, 支持dim倒序 -// note: 使用前须检查 input[dim] == 1 -nvinfer1::ITensor* squeeze_nv_shapetensor(TensorrtEngine* engine, - nvinfer1::ITensor* input, int dim); - -// 对nv tensor进行unsqueeze操作 -nvinfer1::ITensor* unsqueeze_itensor(TensorrtEngine* engine, - nvinfer1::ITensor* input, - const std::vector& axes); - -//TODO: 添加对 nv tensor 进行squeeze操作 -// nvinfer1::ITensor* squeeze_itensor(TensorrtEngine* engine, -// nvinfer1::ITensor* input, -// const std::vector& axes); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/convolution.cpp b/poros/poros/converter/gpu/convolution.cpp deleted file mode 100644 index 179a569e1e2..00000000000 --- a/poros/poros/converter/gpu/convolution.cpp +++ /dev/null @@ -1,271 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file convolution.cpp -* @author tianjinjin@baidu.com -* @date Mon Mar 8 11:36:11 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/convolution.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/converter/gpu/weight.h" -#include "poros/context/poros_global.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -//note: conv?d 与 _convolution 相比,前者入参有7个,后者入参有12个or13个。 -//note2: conv?d 与 _convolution 相比,缺了 transposed 参数(bool类型),默认补零即可。 -//note3: conv?d 与 _convolution 相比,缺了 output_padding 参数(int[]类型),默认也补零即可。 -//note4: conv?d 之间的差异,在于 int[] 的维度不一样。 -bool ConvolutionConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - //basic check - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 12 || inputs.size() == 13 || inputs.size() == 7), - "invaid inputs size for ConvolutionConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for ConvolutionConverter is not Tensor as expected"); - //ATTENTION HERE: assumes weight inputs are all come from prim::Constant and is TensorType. - POROS_CHECK_TRUE((inputs[1]->type()->isSubtypeOf(c10::TensorType::get()) && - inputs[1]->node()->kind() == torch::jit::prim::Constant), - "input[1] for ConvolutionConverter is not Tensor or not come from prim::Constant as expected"); - //ATTENTION HERE: assumes int[] inputs are all come from prim::Constant. - POROS_CHECK_TRUE((inputs[3]->node()->kind() == torch::jit::prim::Constant), - "input[3] for ConvolutionConverter is not come from prim::Constant as expected"); - POROS_CHECK_TRUE((inputs[4]->node()->kind() == torch::jit::prim::Constant), - "input[4] for ConvolutionConverter is not come from prim::Constant as expected"); - POROS_CHECK_TRUE((inputs[5]->node()->kind() == torch::jit::prim::Constant), - "input[5] for ConvolutionConverter is not come from prim::Constant as expected"); - if (inputs.size() == 12 || inputs.size() == 13) { - POROS_CHECK_TRUE((inputs[7]->node()->kind() == torch::jit::prim::Constant), - "input[7] for ConvolutionConverter is not come from prim::Constant as expected"); - } - //extract in - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - - //extract dims settings - auto stride = sizes_to_nvdim((engine->context().get_constant(inputs[3])).toIntList()); - auto padding = sizes_to_nvdim((engine->context().get_constant(inputs[4])).toIntList()); - auto dilation = sizes_to_nvdim((engine->context().get_constant(inputs[5])).toIntList()); - - //handle the difference between _convolution and conv?d. - bool transposed = false; - nvinfer1::Dims out_padding; - int64_t groups = 1; - if (inputs.size() == 12 || inputs.size() == 13) { - transposed = (engine->context().get_constant(inputs[6])).toBool(); - out_padding = sizes_to_nvdim((engine->context().get_constant(inputs[7])).toIntList()); - groups = (engine->context().get_constant(inputs[8])).toInt(); - //situation when conv1d & conv2d & conv3d has no transposed and out_padding paragrams. - } else { - out_padding.nbDims = padding.nbDims; - for (int i = 0; i < padding.nbDims; i++) { - out_padding.d[i] = 0; - } - groups = (engine->context().get_constant(inputs[6])).toInt(); - } - - //handle stride & dilation & padding & out_apdding - if (stride.nbDims == 1) { - stride = unsqueeze_dims(stride, 1, stride.d[0]); - LOG(INFO) << "Reshaped stride for ConvolutionConverter: " << stride; - } - if (dilation.nbDims == 1) { - dilation = unsqueeze_dims(dilation, 1, dilation.d[0]); - LOG(INFO) << "Reshaped dilation for ConvolutionConverter: " << dilation; - } - if (padding.nbDims == 1) { - padding = unsqueeze_dims(padding, 1, 0); - LOG(INFO) << "Reshaped padding for ConvolutionConverter: " << padding; - } - if (out_padding.nbDims == 1) { - out_padding = unsqueeze_dims(out_padding, 1, 0); - LOG(INFO) << "Reshaped out_padding for ConvolutionConverter: " << out_padding; - } - - // According to our tests, when the GPU architecture is Ampere and nvidia tf32 is enabled, - // IConvolutionLayer set PostPadding explicitly even if it is the default value will cause - // tensorrt choose the slow Conv+BN+Relu kernel in some case. - // So we try not to set PostPadding when it is the default value. - bool out_padding_need_set = false; - for (int32_t i = 0; i < out_padding.nbDims; i++) { - if (out_padding.d[i] != 0) { - out_padding_need_set = true; - break; - } - } - - //extract bias - auto maybe_bias = engine->context().get_constant(inputs[2]); - Weights bias; - if (maybe_bias.isTensor()) { - bias = Weights(maybe_bias.toTensor()); - } else {//when bias is None - bias = Weights(); - } - - //extract weight - //the situation can be complex. - //sometimes this params is come from constant. sometimes it's come from another tensor. - //because of the handle strategy we set in prim::Constant. - //we'd better check if it is come from constant first. - auto maybe_weight = engine->context().get_constant(inputs[1]); - /*--------------------------------------------------------------------------- - * when weight is come from constant - ---------------------------------------------------------------------------*/ - if (maybe_weight.isTensor()) { - auto weight = Weights(maybe_weight.toTensor()); - - //first: handle input - auto dims = in->getDimensions(); - auto orig_dims = dims; - POROS_CHECK(orig_dims.nbDims > 2, "Unable to create convolution layer from node: " << *node); - - bool expandDims = (orig_dims.nbDims < 4); - if (expandDims) { - in = add_padding(engine, node, in, 4); - dims = in->getDimensions(); - LOG(INFO) << "Reshaped Input dims: " << dims; - } - - //second: handle dims - if (weight.shape.nbDims < 4) { - for (int i = weight.shape.nbDims; i < 4; ++i) { - weight.shape.d[i] = 1; - } - weight.shape.nbDims = 4; - weight.kernel_shape.nbDims = 2; - weight.kernel_shape.d[1] = 1; - LOG(INFO) << "Reshaped Weights for ConvolutionConverter: " << weight; - } - - //fifth: try to add new layer - nvinfer1::ILayer* new_layer; - if (transposed) { - // shape of deconvolution's weight: [in, out/groups, ...] - auto deconv = engine->network()->addDeconvolutionNd(*in, - weight.shape.d[1] * groups, weight.kernel_shape, weight.data, bias.data); - POROS_CHECK(deconv, "Unable to create deconvolution layer from node: " << *node); - - deconv->setStrideNd(stride); - deconv->setPaddingNd(padding); - deconv->setName((layer_info(node) + "_IDeconvolutionLayer").c_str()); -#if NV_TENSORRT_MAJOR > 7 || (NV_TENSORRT_MAJOR == 7 && NV_TENSORRT_MINOR >= 1) - deconv->setDilationNd(dilation); - deconv->setNbGroups(groups); -#else - POROS_CHECK(groups == 1, "for deconv with groups > 1, require TensorRT version >= 7.1"); - for (int idx = 0; idx < dilation.nbDims; idx++) { - POROS_CHECK(dilation.d[idx] == 1, "for deconv with dilation > 1, require TensorRT version >= 7.1"); - } -#endif - new_layer = deconv; - // when transposed == false - } else { - // shape of convolution's weight: [out, in/groups, ...] - auto conv = engine->network()->addConvolutionNd(*in, - weight.shape.d[0], weight.kernel_shape, weight.data, bias.data); - POROS_CHECK(conv, "Unable to create convolution layer from node: " << *node); - - conv->setStrideNd(stride); - conv->setPaddingMode(nvinfer1::PaddingMode::kCAFFE_ROUND_DOWN); - conv->setPaddingNd(padding); - if (out_padding_need_set) { - conv->setPostPadding(out_padding); - } - conv->setDilationNd(dilation); - conv->setNbGroups(groups); - conv->setName((layer_info(node) + "_IConvolutionLayer").c_str()); - new_layer = conv; - } - - auto out = add_unpadding(engine, node, new_layer->getOutput(0), orig_dims.nbDims); - engine->context().set_tensor(node->outputs()[0], out); - LOG(INFO) << "Output tensor shape: " << out->getDimensions(); - - return true; - /*--------------------------------------------------------------------------- - * when weight is come from other layer's (especial Dequantize layer) output - ---------------------------------------------------------------------------*/ - } else { - auto kernel = engine->context().get_tensor(inputs[1]); - POROS_CHECK_TRUE((kernel != nullptr), "Unable to init input tensor for node: " << *node); - auto kernel_dims = kernel->getDimensions(); - - // Make a new Dims with only the spatial dimensions. - nvinfer1::Dims filter_dim; - int64_t nbSpatialDims = in->getDimensions().nbDims - 2; - POROS_CHECK(nbSpatialDims == (kernel_dims.nbDims - 2), - "Number of input spatial dimensions should match the kernel spatial dimensions"); - filter_dim.nbDims = nbSpatialDims; - filter_dim.d[0] = kernel_dims.d[2]; - filter_dim.d[1] = kernel_dims.d[3]; - - // Initialize a dummy constant kernel to pass it to INetwork->addConvolutionNd/addDeconvolutionNd API. - auto kernel_weights = nvinfer1::Weights{nvinfer1::DataType::kFLOAT, nullptr, 0}; - - nvinfer1::ILayer* layer = nullptr; - if (transposed) { - nvinfer1::IDeconvolutionLayer* deconv = engine->network()->addDeconvolutionNd(*in, - kernel_dims.d[0], - filter_dim, - kernel_weights, - bias.data); - deconv->setStrideNd(stride); - deconv->setDilationNd(dilation); - deconv->setNbGroups(groups); - deconv->setPaddingNd(padding); - // Set deconv kernel weights - deconv->setInput(1, *kernel); - deconv->setName((layer_info(node) + "_IDeconvolutionLayer").c_str()); - POROS_CHECK(deconv, "Unable to create deconv layer with non-const weights from node: " << *node); - layer = deconv; - } else { - nvinfer1::IConvolutionLayer* conv = engine->network()->addConvolutionNd(*in, - kernel_dims.d[0], - filter_dim, - kernel_weights, - bias.data); - conv->setStrideNd(stride); - conv->setPaddingMode(nvinfer1::PaddingMode::kCAFFE_ROUND_DOWN); - conv->setPaddingNd(padding); - if (out_padding_need_set) { - conv->setPostPadding(out_padding); - } - conv->setDilationNd(dilation); - conv->setNbGroups(groups); - // Set conv kernel weights - conv->setInput(1, *kernel); - conv->setName((layer_info(node) + "_IConvolutionLayer").c_str()); - layer = conv; - } - engine->context().set_tensor(node->outputs()[0], layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << layer->getOutput(0)->getDimensions(); - return true; - } -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, ConvolutionConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/convolution.h b/poros/poros/converter/gpu/convolution.h deleted file mode 100644 index 422d1ac0d18..00000000000 --- a/poros/poros/converter/gpu/convolution.h +++ /dev/null @@ -1,66 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file convolution.h -* @author tianjinjin@baidu.com -* @date Wed Aug 11 16:00:26 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class ConvolutionConverter : public GpuConverter { -public: - ConvolutionConverter() {} - virtual ~ConvolutionConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - // bool converter(TensorrtEngine* engine, - // const std::vector inputs, - // const std::vector outputs); - - const std::vector schema_string() { - return {"aten::_convolution(Tensor input, Tensor weight, Tensor? bias, int[] stride, int[] padding, int[] dilation, bool transposed, int[] output_padding, int groups, bool benchmark, bool deterministic, bool cudnn_enabled, bool allow_tf32) -> Tensor", - "aten::_convolution.deprecated(Tensor input, Tensor weight, Tensor? bias, int[] stride, int[] padding, int[] dilation, bool transposed, int[] output_padding, int groups, bool benchmark, bool deterministic, bool cudnn_enabled) -> Tensor", - "aten::conv1d(Tensor input, Tensor weight, Tensor? bias=None, int[1] stride=1, int[1] padding=0, int[1] dilation=1, int groups=1) -> Tensor", - "aten::conv2d(Tensor input, Tensor weight, Tensor? bias=None, int[2] stride=1, int[2] padding=0, int[2] dilation=1, int groups=1) -> Tensor", - "aten::conv3d(Tensor input, Tensor weight, Tensor? bias=None, int[3] stride=1, int[3] padding=0, int[3] dilation=1, int groups=1) -> Tensor" - }; - } - - const std::vector node_kind() { - return {torch::jit::aten::_convolution, - torch::jit::aten::conv1d, - torch::jit::aten::conv2d, - torch::jit::aten::conv3d}; - } -}; - - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/einsum.cpp b/poros/poros/converter/gpu/einsum.cpp deleted file mode 100644 index 9945aa54fc7..00000000000 --- a/poros/poros/converter/gpu/einsum.cpp +++ /dev/null @@ -1,72 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file einsum.cpp -* @author tianshaoqing@baidu.com -* @date Wed Jul 06 11:24:51 CST 2022 -* @brief -**/ - -#include "poros/converter/gpu/einsum.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -// aten::einsum(str equation, Tensor[] tensors) -> (Tensor) -bool EinsumConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for EinsumConverter"); - POROS_CHECK_TRUE(inputs[1]->type()->isSubtypeOf(c10::ListType::ofTensors()), - "input[1] for EinsumConverter is not TensorList as expected."); - - // extract equation string - torch::jit::IValue equation_ivalue = engine->context().get_constant(inputs[0]); - POROS_CHECK_TRUE(equation_ivalue.isString(), "EinsumConverter input[0] is not constant string as expected."); - std::string equation_str = equation_ivalue.toStringRef(); - - // 大写转小写,nvinfer1::IEinsumLayer不支持equation中包含大写字母 - for (auto it = equation_str.begin(); it != equation_str.end(); it++) { - if ((*it) >= 'A' && (*it) <= 'Z') { - *it = *it + 32; - } - } - - // extract tensorlist - // mark:单测时输入2个以上的tensor trt会报错 nbInputs > 0 && nbInputs <= MAX_EINSUM_NB_INPUTS - // 不确定MAX_EINSUM_NB_INPUTS是固定=2还是根据equation来定,暂时不加判断。 - std::vector tensorlist; - POROS_CHECK_TRUE(engine->context().get_tensorlist(inputs[1], tensorlist), "EinsumConverter " - "extract tensor list error."); - - nvinfer1::IEinsumLayer* einsum_layer = engine->network()->addEinsum(tensorlist.data(), - tensorlist.size(), - equation_str.c_str()); - einsum_layer->setName((layer_info(node) + "_IEinsumLayer").c_str()); - - nvinfer1::ITensor* output = einsum_layer->getOutput(0); - - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output shape: " << output->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, EinsumConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/einsum.h b/poros/poros/converter/gpu/einsum.h deleted file mode 100644 index f174cd2ad36..00000000000 --- a/poros/poros/converter/gpu/einsum.h +++ /dev/null @@ -1,57 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file einsum.h -* @author tianshaoqing@baidu.com -* @date Wed Jul 06 11:24:51 CST 2022 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class EinsumConverter : public GpuConverter { -public: - EinsumConverter() {} - virtual ~EinsumConverter() {} - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - const std::vector schema_string() { - return {"aten::einsum(str equation, Tensor[] tensors) -> (Tensor)"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::einsum}; - } - - // mark: einsum动态的规则比较复杂,是根据equation情况来定的,先禁止dy - bool assign_schema_attr() { - return assign_schema_attr_helper({{"aten::einsum(str equation, Tensor[] tensors) -> (Tensor)", {0, 0}}}); - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/element_wise.cpp b/poros/poros/converter/gpu/element_wise.cpp deleted file mode 100644 index 15b7b8f504a..00000000000 --- a/poros/poros/converter/gpu/element_wise.cpp +++ /dev/null @@ -1,399 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file element_wise.cpp -* @author tianjinjin@baidu.com -* @date Fri Aug 27 15:32:36 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/element_wise.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -nvinfer1::ITensor* GreaterOrLessConverter::scalar_to_nvtensor(TensorrtEngine* engine, at::Scalar s) { - nvinfer1::ITensor* out; - if (s.isIntegral(false)) { - auto s_int = s.to(); - auto s_t = torch::tensor({s_int}).to(at::kInt); - out = tensor_to_const(engine, s_t); - } else if (s.isBoolean()) { - auto s_bool = s.to(); - auto s_t = torch::tensor({s_bool}).to(at::kBool); - out = tensor_to_const(engine, s_t); - } else if (s.isFloatingPoint()) { - auto other_float = s.to(); - auto s_t = torch::tensor({other_float}); - out = tensor_to_const(engine, s_t); - } else { - out = nullptr; - POROS_THROW_ERROR("Unsupported data type for scalar. Found: (" << s.type() << ")"); - } - return out; -} - -/* -"aten::gt.Tensor(Tensor self, Tensor other) -> Tensor", -"aten::gt.Scalar(Tensor self, Scalar other) -> Tensor",*/ -bool GreaterOrLessConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for GreaterOrLessConverter"); - - int schema_index = 0; - for (std::string schema : this->schema_string()) { - if (node->schema().operator_name() == torch::jit::parseSchema(schema).operator_name()) { - break; - } - schema_index++; - } - - if (schema_index >= 8 && schema_index <= 11) { - int a = engine->context().get_constant(inputs[0]).toScalar().to(); - int b = engine->context().get_constant(inputs[1]).toScalar().to(); - bool output = true; - if (schema_index == 8) { - output = (a > b); - } - if (schema_index == 9) { - output = (a < b); - } - if (schema_index == 10) { - output = (a >= b); - } - if (schema_index == 11) { - output = (a <= b); - } - engine->context().set_constant(node->outputs()[0], output); - return true; - } - - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for GreaterOrLessConverter is not Tensor as expected"); - - //extract self - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - auto other = engine->context().get_tensor(inputs[1]); - //when other input is Scalar - if (other == nullptr) { - auto other_const = engine->context().get_constant(inputs[1]); - if (other_const.isScalar()) { - other = scalar_to_nvtensor(engine, other_const.toScalar()); - if (self->getType() != other->getType()) { - other = cast_itensor(engine, other, self->getType()); - } - } else { - POROS_THROW_ERROR("Unable to get input other value for GreaterOrLessConverter"); - } - } - - nvinfer1::ElementWiseOperation ew_option; - std::string name_suffix; - if (node->kind() == torch::jit::aten::gt || node->kind() == torch::jit::aten::ge) { - ew_option = nvinfer1::ElementWiseOperation::kGREATER; - name_suffix = "_greater"; - } else if (node->kind() == torch::jit::aten::lt || node->kind() == torch::jit::aten::le) { - ew_option = nvinfer1::ElementWiseOperation::kLESS; - name_suffix = "_less"; - } else { - POROS_THROW_ERROR("Meet some unknown node kind in GreaterOrLessConverter"); - } - - auto new_layer = add_elementwise(engine, - ew_option, - self, - other, - layer_info(node) + name_suffix); - POROS_CHECK(new_layer, "Unable to create element wise layer from node: " << *node); - - //situation: aten::gt or aten::lt - if (node->kind() == torch::jit::aten::gt || node->kind() == torch::jit::aten::lt) { - engine->context().set_tensor(node->outputs()[0], new_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << new_layer->getOutput(0)->getDimensions(); - return true; - } - - // situation: aten::ge or aten::le, - // we should set three layers: kGREATER(or kLESS) and kEQUAL and kOR. - if (node->kind() == torch::jit::aten::ge || node->kind() == torch::jit::aten::le) { - //equal layer - auto equal = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kEQUAL, - self, - other, - layer_info(node) + "_equal"); - POROS_CHECK(equal, "Unable to create Equal layer from node: " << *node); - - //or layer - auto or_op = engine->network()->addElementWise( - *new_layer->getOutput(0), - *equal->getOutput(0), - nvinfer1::ElementWiseOperation::kOR); - POROS_CHECK(or_op, "Unable to create Or layer from node: " << *node); - - or_op->setName((layer_info(node) + "_or").c_str()); - engine->context().set_tensor(node->outputs()[0], or_op->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << or_op->getOutput(0)->getDimensions(); - return true; - } - - POROS_THROW_ERROR("Meet some unknown node kind in GreaterOrLessConverter"); - return false; -} - -/* -"aten::eq.Tensor(Tensor self, Tensor other) -> Tensor", -"aten::eq.Scalar(Tensor self, Scalar other) -> Tensor",*/ -bool EqualOrNotequalConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for EqualOrNotequalConverter"); - - int schema_index = 0; - for (std::string schema : this->schema_string()) { - if (node->schema().operator_name() == torch::jit::parseSchema(schema).operator_name()) { - break; - } - schema_index++; - } - - if (schema_index == 4 || schema_index == 5) { - int a = engine->context().get_constant(inputs[0]).toScalar().to(); - int b = engine->context().get_constant(inputs[1]).toScalar().to(); - bool output = true; - if (schema_index == 4) { - output = (a == b); - } - if (schema_index == 5) { - output = (a != b); - } - engine->context().set_constant(node->outputs()[0], output); - return true; - } - - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for EqualOrNotequalConverter is not Tensor as expected"); - - //extract self - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - auto other = engine->context().get_tensor(inputs[1]); - //when other input is Scalar - if (other == nullptr) { - auto other_const = engine->context().get_constant(inputs[1]); - if (other_const.isScalar()) { - auto other_scalar = other_const.toScalar().to(); - other = tensor_to_const(engine, torch::tensor({other_scalar})); - if (node->kind() == torch::jit::aten::eq) { - //TODO: when aten::ne situation, we may alse need to cat functions below?? - if (self->getType() == nvinfer1::DataType::kBOOL) { - if (other_scalar == 0 || other_scalar == 1) { - LOG(INFO) << "Since input tensor is type bool, casting input tensor and scalar to int32"; - other = cast_itensor(engine, other, nvinfer1::DataType::kINT32); - self = cast_itensor(engine, self, nvinfer1::DataType::kINT32); - } else { - LOG(WARNING) << "Input Tensor has type bool, but scalar is not 0 or 1. Found: " << other_scalar; - return false; - } - } - if (self->getType() != other->getType()) { - other = cast_itensor(engine, other, self->getType()); - } - } - } else { - POROS_THROW_ERROR("Unable to get input other value for EqualOrNotequalConverter"); - } - } - - auto equal_layer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kEQUAL, - self, - other, - layer_info(node) + "_equal"); - POROS_CHECK(equal_layer, "Unable to create equal layer from node: " << *node); - - //situation: aten::eq - if (node->kind() == torch::jit::aten::eq) { - engine->context().set_tensor(node->outputs()[0], equal_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << equal_layer->getOutput(0)->getDimensions(); - return true; - } - - // situation: aten::ne - // we should set another two layers: all-ones layer and kXOR. - if (node->kind() == torch::jit::aten::ne) { - // XOR with ones negates and produces not_equal result - auto options = torch::TensorOptions().dtype(torch::kFloat32); - auto ones = at::full({1}, 1, {options}); - auto ones_tensor = tensor_to_const(engine, ones); - nvinfer1::IIdentityLayer* cast_layer = engine->network()->addIdentity(*ones_tensor); - cast_layer->setName((layer_info(node) + "_IIdentityLayer").c_str()); - cast_layer->setOutputType(0, nvinfer1::DataType::kBOOL); - - //xor layer - auto xor_op = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kXOR, - cast_layer->getOutput(0), - equal_layer->getOutput(0), - layer_info(node) + "_xor"); - POROS_CHECK(xor_op, "Unable to create ne (not equal) layer from node: " << *node); - engine->context().set_tensor(node->outputs()[0], xor_op->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << xor_op->getOutput(0)->getDimensions(); - return true; - } - - POROS_THROW_ERROR("Meet some unknown node kind in EqualOrNotequalConverter"); - return false; -} - -/* -"aten::pow.Tensor_Tensor(Tensor self, Tensor exponent) -> Tensor", -"aten::pow.Tensor_Scalar(Tensor self, Scalar exponent) -> Tensor",*/ -bool PowOrFloordivideConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for PowOrFloordivideConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for PowOrFloordivideConverter is not Tensor as expected"); - - //extract self - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - auto other = engine->context().get_tensor(inputs[1]); - //when other input is Scalar - if (other == nullptr) { - auto other_const = engine->context().get_constant(inputs[1]); - if (other_const.isScalar()) { - auto other_scalar = other_const.toScalar().to(); - other = tensor_to_const(engine, torch::tensor({other_scalar})); - } else { - POROS_THROW_ERROR("Unable to get input other value for PowOrFloordivideConverter"); - } - } - - nvinfer1::ElementWiseOperation ew_option; - std::string name_suffix; - if (node->kind() == torch::jit::aten::pow) { - ew_option = nvinfer1::ElementWiseOperation::kPOW; - name_suffix = "_pow"; - //TODO: handle floor_divide situaition - // } else if (node->kind() == torch::jit::at::floor_divide) { - // ew_option = nvinfer1::ElementWiseOperation::kFLOOR_DIV; - // name_suffix = "_floor_div"; - } else { - POROS_THROW_ERROR("Meet some unknown node kind in PowOrFloordivideConverter"); - } - - auto new_layer = add_elementwise(engine, - ew_option, - self, - other, - layer_info(node) + name_suffix); - POROS_CHECK(new_layer, "Unable to create pow or floor_divide layer from node: " << *node); - engine->context().set_tensor(node->outputs()[0], new_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << new_layer->getOutput(0)->getDimensions(); - return true; -} - -/* -"aten::clamp(Tensor self, Scalar? min=None, Scalar? max=None) -> Tensor", -"aten::clamp_min(Tensor self, Scalar min) -> Tensor", -"aten::clamp_max(Tensor self, Scalar max) -> Tensor",*/ -bool ClampConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2 || inputs.size() == 3), "invaid inputs size for ClampConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for ClampConverter is not Tensor as expected"); - - //extract self - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - auto clamp_layer_out = self; - - torch::jit::IValue maybe_min; - torch::jit::IValue maybe_max; - if (node->kind() == torch::jit::aten::clamp) { - maybe_min = engine->context().get_constant(inputs[1]); - maybe_max = engine->context().get_constant(inputs[2]); - } else if (node->kind() == torch::jit::aten::clamp_min) { - maybe_min = engine->context().get_constant(inputs[1]); - maybe_max = torch::jit::IValue(); - } else { //node->kind() == torch::jit::aten::clamp_max - maybe_min = torch::jit::IValue(); - maybe_max = engine->context().get_constant(inputs[1]); - } - - if (maybe_min.isScalar() && maybe_max.isScalar()) { - // note: same as pytorch, first max, then min - auto limit = maybe_min.toScalar().to(); - auto limit_tensor = tensor_to_const(engine, torch::tensor({limit})); - auto limit_layer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kMAX, - self, - limit_tensor, - layer_info(node) + "_max"); - POROS_CHECK(limit_layer, "Unable to create elementwise(KMAX) layer for node: " << *node); - clamp_layer_out = limit_layer->getOutput(0); - limit = maybe_max.toScalar().to(); - limit_tensor = tensor_to_const(engine, torch::tensor({limit})); - limit_layer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kMIN, - clamp_layer_out, - limit_tensor, - layer_info(node) + "_min"); - POROS_CHECK(limit_layer, "Unable to create elementwise(KMIN) layer for node: " << *node); - clamp_layer_out = limit_layer->getOutput(0); - } else if (maybe_min.isScalar()) { - auto limit = maybe_min.toScalar().to(); - auto limit_tensor = tensor_to_const(engine, torch::tensor({limit})); - auto limit_layer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kMAX, - self, - limit_tensor, - layer_info(node) + "_max"); - POROS_CHECK(limit_layer, "Unable to create elementwise(KMAX) layer for node: " << *node); - clamp_layer_out = limit_layer->getOutput(0); - } else if (maybe_max.isScalar()) { - auto limit = maybe_max.toScalar().to(); - auto limit_tensor = tensor_to_const(engine, torch::tensor({limit})); - auto limit_layer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kMIN, - self, - limit_tensor, - layer_info(node) + "_min"); - POROS_CHECK(limit_layer, "Unable to create elementwise(KMIN) layer for node: " << *node); - clamp_layer_out = limit_layer->getOutput(0); - } - - engine->context().set_tensor(node->outputs()[0], clamp_layer_out); - LOG(INFO) << "Output tensor shape: " << clamp_layer_out->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, GreaterOrLessConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, EqualOrNotequalConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, PowOrFloordivideConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, ClampConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/element_wise.h b/poros/poros/converter/gpu/element_wise.h deleted file mode 100644 index 6851063085b..00000000000 --- a/poros/poros/converter/gpu/element_wise.h +++ /dev/null @@ -1,181 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file element_wise.h -* @author tianjinjin@baidu.com -* @date Fri Aug 27 15:32:36 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class GreaterOrLessConverter : public GpuConverter { -public: - GreaterOrLessConverter() {} - virtual ~GreaterOrLessConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - // note: gt.int, lt.int, ge.int le.int maybe should not support better. - // because they usually appear with if or loop. - const std::vector schema_string() { - return {"aten::gt.Tensor(Tensor self, Tensor other) -> Tensor", - "aten::gt.Scalar(Tensor self, Scalar other) -> Tensor", - "aten::lt.Tensor(Tensor self, Tensor other) -> Tensor", - "aten::lt.Scalar(Tensor self, Scalar other) -> Tensor", - "aten::ge.Tensor(Tensor self, Tensor other) -> Tensor", - "aten::ge.Scalar(Tensor self, Scalar other) -> Tensor", - "aten::le.Tensor(Tensor self, Tensor other) -> Tensor", - "aten::le.Scalar(Tensor self, Scalar other) -> Tensor", - "aten::gt.int(int a, int b) -> (bool)", - "aten::lt.int(int a, int b) -> (bool)", - "aten::ge.int(int a, int b) -> (bool)", - "aten::le.int(int a, int b) -> (bool)", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::gt.Tensor_out(Tensor self, Tensor other, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::lt.Tensor_out(Tensor self, Tensor other, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::ge.Tensor_out(Tensor self, Tensor other, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::le.Tensor_out(Tensor self, Tensor other, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::gt.Scalar_out(Tensor self, Scalar other, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::lt.Scalar_out(Tensor self, Scalar other, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::ge.Scalar_out(Tensor self, Scalar other, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::le.Scalar_out(Tensor self, Scalar other, *, Tensor(a!) out) -> Tensor(a!)", - * **/ - const std::vector node_kind() { - return {torch::jit::aten::gt, - torch::jit::aten::lt, - torch::jit::aten::ge, - torch::jit::aten::le}; - } - -private: - nvinfer1::ITensor* scalar_to_nvtensor(TensorrtEngine* engine, at::Scalar s); -}; - -class EqualOrNotequalConverter : public GpuConverter { -public: - EqualOrNotequalConverter() {} - virtual ~EqualOrNotequalConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - // note: eq.int, ne.int maybe should not support better. - // because they usually appear with if or loop. - const std::vector schema_string() { - return {"aten::eq.Tensor(Tensor self, Tensor other) -> Tensor", - "aten::eq.Scalar(Tensor self, Scalar other) -> Tensor", - "aten::ne.Tensor(Tensor self, Tensor other) -> Tensor", - "aten::ne.Scalar(Tensor self, Scalar other) -> Tensor", - "aten::eq.int(int a, int b) -> (bool)", - "aten::ne.int(int a, int b) -> (bool)" - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::eq.Tensor_out(Tensor self, Tensor other, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::ne.Tensor_out(Tensor self, Tensor other, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::eq.Scalar_out(Tensor self, Scalar other, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::ne.Scalar_out(Tensor self, Scalar other, *, Tensor(a!) out) -> Tensor(a!)", - * **/ - const std::vector node_kind() { - return {torch::jit::aten::eq, - torch::jit::aten::ne, - }; - } -}; - -class PowOrFloordivideConverter : public GpuConverter { -public: - PowOrFloordivideConverter() {} - virtual ~PowOrFloordivideConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::pow.Tensor_Tensor(Tensor self, Tensor exponent) -> Tensor", - "aten::pow.Tensor_Scalar(Tensor self, Scalar exponent) -> Tensor", - //"aten::floor_divide(Tensor self, Tensor other) -> Tensor", - //"aten::floor_divide.Scalar(Tensor self, Scalar other) -> Tensor", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::pow.Tensor_Tensor_out(Tensor self, Tensor exponent, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::pow.Tensor_Scalar_out(Tensor self, Scalar exponent, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::pow.Scalar_out(Scalar self, Tensor exponent, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::pow.Scalar(Scalar self, Tensor exponent) -> Tensor", - * - * aten::floor_divide.out(Tensor self, Tensor other, *, Tensor(a!) out) -> Tensor(a!) - * **/ - const std::vector node_kind() { - return {torch::jit::aten::pow, - //torch::jit::aten::floor_divide, - }; - } -}; - -class ClampConverter : public GpuConverter { -public: - ClampConverter() {} - virtual ~ClampConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::clamp(Tensor self, Scalar? min=None, Scalar? max=None) -> Tensor", - "aten::clamp_min(Tensor self, Scalar min) -> Tensor", - "aten::clamp_max(Tensor self, Scalar max) -> Tensor", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::clamp.Tensor(Tensor self, Tensor? min=None, Tensor? max=None) -> Tensor", - * "aten::clamp_min.Tensor(Tensor self, Tensor min) -> Tensor", - * "aten::clamp_max.Tensor(Tensor self, Tensor max) -> Tensor", - * - * "aten::clamp.out(Tensor self, Scalar? min=None, Scalar? max=None, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::clamp_min.out(Tensor self, Scalar min, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::clamp_max.out(Tensor self, Scalar max, *, Tensor(a!) out) -> Tensor(a!)", - * - * "aten::clamp.Tensor_out(Tensor self, Tensor? min=None, Tensor? max=None, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::clamp_min.Tensor_out(Tensor self, Tensor min, *, Tensor(a!) out) -> Tensor(a!)" - * "aten::clamp_max.Tensor_out(Tensor self, Tensor max, *, Tensor(a!) out) -> Tensor(a!)" - * **/ - const std::vector node_kind() { - return {torch::jit::aten::clamp, - torch::jit::aten::clamp_min, - torch::jit::aten::clamp_max, - }; - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/expand.cpp b/poros/poros/converter/gpu/expand.cpp deleted file mode 100644 index 7606fe1d792..00000000000 --- a/poros/poros/converter/gpu/expand.cpp +++ /dev/null @@ -1,315 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file expand.cpp -* @author tianjinjin@baidu.com -* @date Mon Mar 8 11:36:11 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/expand.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/* -"aten::expand(Tensor(a) self, int[] size, *, bool implicit=False) -> Tensor(a)", -"aten::expand_as(Tensor(a) self, Tensor other) -> Tensor(a)"*/ -bool ExpandConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 3 || inputs.size() == 2), "invaid inputs size for ExpandConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for ExpandConverter is not Tensor as expected"); - if (inputs.size() == 3) { - POROS_CHECK_TRUE((inputs[2]->node()->kind() == torch::jit::prim::Constant), - "input[2] for ExpandConverter is not come from prim::Constant as expected"); - } - //extract in - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - auto input_dims = in->getDimensions(); - auto input_rank = in->getDimensions().nbDims; - - //extract target_dims & init expanded_dims_tensor - nvinfer1::Dims target_dims; - nvinfer1::ITensor* expanded_dims_tensor = nullptr; - bool is_expand_layer = false; - bool has_tensor_scalar = false; - if (node->kind() == torch::jit::aten::expand) { - has_tensor_scalar = check_inputs_tensor_scalar(engine, node); - if (has_tensor_scalar) { - expanded_dims_tensor = get_tensor_scalar(inputs[1]); - POROS_CHECK_TRUE((expanded_dims_tensor != nullptr), node_info(node) + std::string("get int nvtensor false.")); - target_dims = expanded_dims_tensor->getDimensions(); - } else { - auto expanded_size = (engine->context().get_constant(inputs[1])).toIntList(); - target_dims = sizes_to_nvdim(expanded_size); - auto expanded_size_tensor = torch::tensor(expanded_size.vec(), torch::kInt32); - expanded_dims_tensor = tensor_to_const(engine, expanded_size_tensor); - } - is_expand_layer = true; - } else { //node->kind() == torch::jit::aten::expand_as - auto target_tensor = engine->context().get_tensor(inputs[1]); - target_dims = target_tensor->getDimensions(); - expanded_dims_tensor = engine->network()->addShape(*target_tensor)->getOutput(0); - } - auto output_rank = target_dims.nbDims; - if (has_tensor_scalar) { - output_rank = target_dims.d[0]; - } - - POROS_CHECK(input_rank <= output_rank, - "Number of dimensions of the desired expansion must be greater than or equal to the number of input dimensions"); - - auto is_dynamic_shape = PorosGlobalContext::instance().get_poros_options().is_dynamic; - - //situation1: ---------- when input is dynamic shape ------------- - if (is_dynamic_shape) { - // Validate the expansion. Eg: an input of [3, 1] can be expanded to [1, 3, 4] but not [3, 4, 1] - if (!has_tensor_scalar) { - for (int64_t i = target_dims.nbDims - 1; i >= 0; --i) { - int64_t offset = target_dims.nbDims - 1 - i; - int64_t dim = input_dims.nbDims - 1 - offset; - int64_t size = (dim >= 0) ? input_dims.d[dim] : 1; - int64_t target_size = target_dims.d[i]; - // Passing -1 as the size for a dimension means not changing the size of that dimension in expand layer. - if (target_size != -1) { - if (size != target_size) { - // if size == -1, we can't validate the expansion before setBindingDimensions. - POROS_CHECK_TRUE((size == -1 || size == 1), "The expanded size of tensor (" << std::to_string(target_size) << ")" - << " must match the existing size (" << std::to_string(size) << ")" << " at dimension " << i); - } - } else { - //expand 的 target_size 不可以出现-1,(因为是intlist),但expand_as 可以,因为通过shape获取真实的size。 - POROS_CHECK_TRUE(!(is_expand_layer && dim < 0), "The target dims " << target_dims << " for node [" - << node_info(node) << "] is illegal, should not have -1 value"); - } - } - } else { - LOG(INFO) << "aten::expend ints tensor maybe not right, because has no check."; - } - - size_t max_rank = std::max(input_rank, output_rank); - // Dimensions are right alignment. Eg: an input of [3, 1] and max_rank = 4, the result of concat is [1, 1, 3, 1] - nvinfer1::ITensor* new_input_shape_tensor = nullptr; - if (max_rank - input_rank > 0) { - torch::Tensor the_one = torch::tensor(std::vector(max_rank - input_rank, 1), torch::kInt32); - auto one_tensor = tensor_to_const(engine, the_one); - auto in_shape_tensor = engine->network()->addShape(*in)->getOutput(0); - nvinfer1::ITensor* const args[2] = {one_tensor, in_shape_tensor}; - new_input_shape_tensor = engine->network()->addConcatenation(args, 2)->getOutput(0); - } else { //max_rank - input_rank == 0 - new_input_shape_tensor = engine->network()->addShape(*in)->getOutput(0); - } - auto new_output_shape_tensor = expanded_dims_tensor; - - // Add a reshape layer to expand dims - auto shuffle = engine->network()->addShuffle(*in); - shuffle->setInput(1, *new_input_shape_tensor); - shuffle->setName((layer_info(node) + "_IShuffleLayer").c_str()); - - // Start the slicing from beginning of tensor since this is an expand layer - std::vector start_vec(max_rank, 0); - nvinfer1::Dims starts_dim = sizes_to_nvdim(c10::IntArrayRef(start_vec)); - at::Tensor th_start = torch::tensor(nvdim_to_sizes(starts_dim), torch::kInt32); - auto starts = tensor_to_const(engine, th_start); - - // compute sizes = max(x,y). - auto sizes = engine->network()->addElementWise(*new_input_shape_tensor, - *new_output_shape_tensor, - nvinfer1::ElementWiseOperation::kMAX)->getOutput(0); - nvinfer1::Dims sizes_dim{-1, {}}; - sizes_dim.nbDims = max_rank; - - // Compute (x > 1 ? 1 : 0) for x in newDims, assuming positive x, using only TensorRT operations. - // min(1, sub(input_shape, 1)) - torch::Tensor thOne = torch::tensor({1}, torch::kInt32); - auto thone_tensor = tensor_to_const(engine, thOne); - auto x_sub_one = engine->network()->addElementWise(*new_input_shape_tensor, - *thone_tensor, - nvinfer1::ElementWiseOperation::kSUB)->getOutput(0); - auto strides = engine->network()->addElementWise(*thone_tensor, - *x_sub_one, - nvinfer1::ElementWiseOperation::kMIN)->getOutput(0); - nvinfer1::Dims strides_dim{-1, {}}; - strides_dim.nbDims = max_rank; - - // Slice layer does the expansion in TRT. Desired output size is specified by sizes input at index 2. - auto slice = engine->network()->addSlice(*shuffle->getOutput(0), starts_dim, sizes_dim, strides_dim); - slice->setInput(1, *starts); - slice->setInput(2, *sizes); - slice->setInput(3, *strides); - slice->setName((layer_info(node) + "_ISliceLayer").c_str()); - - engine->context().set_tensor(node->outputs()[0], slice->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << slice->getOutput(0)->getDimensions(); - return true; - - //situation2: ---------- when input is NOT dynamic shape ------------- - } else { - // Validate the expansion. Eg: an input of [3, 1] can be expanded to [1, 3, 4] but not [3, 4, 1] - for (int64_t i = target_dims.nbDims - 1; i >= 0; --i) { - int64_t offset = target_dims.nbDims - 1 - i; - int64_t dim = input_dims.nbDims - 1 - offset; - int64_t size = (dim >= 0) ? input_dims.d[dim] : 1; - int64_t target_size = target_dims.d[i]; - // In expand layer passing -1 as the size for a dimension means not changing the size of that dimension. - if (target_size != -1) { - if (size != target_size) { - POROS_CHECK_TRUE((size == 1), "The expanded size of tensor (" << std::to_string(target_size) << ")" - << " must match the existing size (" << std::to_string(size) << ")" << " at dimension " << i); - } - } else { - //target_size 不可以出现 -1. - POROS_CHECK_TRUE((dim >= 0), "The target dims " << target_dims << " for node [" - << node_info(node) << "] is illegal, should not have -1 value"); - // in(3, 1), expand(3, -1, 4) -> expand(3, 3, 4) - target_dims.d[i] = input_dims.d[dim]; - } - } - - auto num_expand_dims = target_dims.nbDims - input_dims.nbDims; - if (num_expand_dims > 0) { - nvinfer1::Dims reshape_dims; - reshape_dims.nbDims = target_dims.nbDims; - for (int64_t i = 0; i < num_expand_dims; i++) { - reshape_dims.d[i] = 1; - } - for (int64_t i = 0; i < input_dims.nbDims; i++) { - reshape_dims.d[num_expand_dims + i] = input_dims.d[i]; - } - - // Add a reshape layer to expand dims - auto reshape_layer = engine->network()->addShuffle(*in); - reshape_layer->setReshapeDimensions(reshape_dims); - reshape_layer->setName((layer_info(node) + "_IShuffleLayer").c_str()); - in = reshape_layer->getOutput(0); - LOG(INFO) << "Input reshaped to : " << in->getDimensions() << " from " << input_dims; - } - - // Start the slicing from beginning of tensor since this is an expand layer - std::vector start_vec(target_dims.nbDims, 0); - auto start_offset = sizes_to_nvdim(c10::IntArrayRef(start_vec)); - - // Set the stride of non singleton dimension to 1 - std::vector strides_vec(target_dims.nbDims, 0); - for (int64_t i = 0; i < target_dims.nbDims; i++) { - strides_vec[i] = (in->getDimensions().d[i] != 1); - } - - auto strides = sizes_to_nvdim(c10::IntArrayRef(strides_vec)); - // Slice layer does the expansion in TRT. Desired output size is specified by target_dims - auto slice_layer = engine->network()->addSlice(*in, start_offset, target_dims, strides); - slice_layer->setName((layer_info(node) + "_ISliceLayer").c_str()); - engine->context().set_tensor(node->outputs()[0], slice_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << slice_layer->getOutput(0)->getDimensions(); - return true; - } -} - -/* -"aten::repeat(Tensor self, int[] repeats) -> Tensor", -*/ -bool RepeatConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for RepeatConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for RepeatConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[1]->node()->kind() == torch::jit::prim::Constant), - "input[2] for RepeatConverter is not come from prim::Constant as expected"); - - //extract in - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - auto input_dims = in->getDimensions(); - int input_rank = input_dims.nbDims; - - //extract repeats - auto repeats = (engine->context().get_constant(inputs[1])).toIntList().vec(); - int repeats_rank = repeats.size(); - - POROS_CHECK(repeats_rank >= input_rank, "Number of repeat dimensions cannot be smaller than number of input dimensions"); - auto num_expand_dims = repeats_rank - input_rank; - - auto is_dynamic_shape = PorosGlobalContext::instance().get_poros_options().is_dynamic; - if (is_dynamic_shape) { - nvinfer1::ITensor* new_input_shape_tensor; - if (num_expand_dims > 0) { - torch::Tensor the_one = torch::tensor(std::vector(num_expand_dims, 1), torch::kInt32); - auto one_tensor = tensor_to_const(engine, the_one); - auto in_shape_tensor = engine->network()->addShape(*in)->getOutput(0); - nvinfer1::ITensor* const args[2] = {one_tensor, in_shape_tensor}; - new_input_shape_tensor = engine->network()->addConcatenation(args, 2)->getOutput(0); - } else { //num_expand_dims == 0 - new_input_shape_tensor = engine->network()->addShape(*in)->getOutput(0); - } - - // Add a reshape layer to expand dims - auto shuffle = engine->network()->addShuffle(*in); - shuffle->setInput(1, *new_input_shape_tensor); - shuffle->setName((layer_info(node) + "_IShuffleLayer").c_str()); - in = shuffle->getOutput(0); - } else { - if (num_expand_dims > 0) { - nvinfer1::Dims reshape_dims; - reshape_dims.nbDims = repeats.size(); - for (int i = 0; i < num_expand_dims; i++) { - reshape_dims.d[i] = 1; - } - for (int i = 0; i < input_rank; i++) { - reshape_dims.d[num_expand_dims + i] = input_dims.d[i]; - } - - // Add a reshape layer to expand dims - auto reshape_layer = engine->network()->addShuffle(*in); - reshape_layer->setReshapeDimensions(reshape_dims); - reshape_layer->setName((layer_info(node) + "_IShuffleLayer").c_str()); - in = reshape_layer->getOutput(0); - LOG(INFO) << "Input reshaped to : " << in->getDimensions() << " from " << input_dims; - } - } - - // Concat across all repeat axes. - // TODO: Implementation might not be performant. Explore other strategies to improve performance. - for (int i = repeats.size() - 1; i >= 0; --i) { - std::vector tensors_vec; - for (int j = 0; j < repeats[i]; j++) { - tensors_vec.push_back(in); - } - auto concat_layer = engine->network()->addConcatenation(tensors_vec.data(), tensors_vec.size()); - concat_layer->setAxis(i); - concat_layer->setName((layer_info(node) + "_IConcatenationLayer_" + std::to_string(i)).c_str()); - in = concat_layer->getOutput(0); - } - - engine->context().set_tensor(node->outputs()[0], in); - LOG(INFO) << "Output tensor shape: " << in->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, ExpandConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, RepeatConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/expand.h b/poros/poros/converter/gpu/expand.h deleted file mode 100644 index f40f84078a0..00000000000 --- a/poros/poros/converter/gpu/expand.h +++ /dev/null @@ -1,76 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file expand.h -* @author tianjinjin@baidu.com -* @date Mon Aug 16 12:26:28 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class ExpandConverter : public GpuConverter { -public: - ExpandConverter() {} - virtual ~ExpandConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::expand(Tensor(a) self, int[] size, *, bool implicit=False) -> Tensor(a)", - "aten::expand_as(Tensor(a) self, Tensor other) -> Tensor(a)",}; - } - - const std::vector node_kind() { - return {torch::jit::aten::expand, - torch::jit::aten::expand_as}; - } - - bool assign_schema_attr() { - return assign_schema_attr_helper({{"aten::expand(Tensor(a) self, int[] size, *, bool implicit=False) -> Tensor(a)", {1, 1}}}); - } -}; - -class RepeatConverter : public GpuConverter { -public: - RepeatConverter() {} - virtual ~RepeatConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::repeat(Tensor self, int[] repeats) -> Tensor"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::repeat}; - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/generate.cpp b/poros/poros/converter/gpu/generate.cpp deleted file mode 100644 index 61942a2aa84..00000000000 --- a/poros/poros/converter/gpu/generate.cpp +++ /dev/null @@ -1,578 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file generate.cpp -* @author tianshaoqing@baidu.com -* @date Mon Dec 6 14:29:20 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/generate.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -// aten::zeros_like(Tensor self, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None, MemoryFormat? memory_format=None) -> Tensor -bool ZerosLikeConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 6), "invaid inputs size for ZerosLikeConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for ZerosLikeConverter is not Tensor as expected"); - // extract self - nvinfer1::ITensor* self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - torch::jit::IValue maybe_type = engine->context().get_constant(inputs[1]); - if (maybe_type.isNone()) { - nvinfer1::ILayer* sub_layer = add_elementwise(engine, nvinfer1::ElementWiseOperation::kSUB, - self, self, layer_info(node) + "_sub"); - nvinfer1::ITensor* output = sub_layer->getOutput(0); - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - return true; - } else { - nvinfer1::ITensor* shape_tensor = engine->network()->addShape(*self)->getOutput(0); - int32_t self_rank = (shape_tensor->getDimensions()).d[0]; - - at::ScalarType input_type = maybe_type.toScalarType(); - nvinfer1::IFillLayer* fill_layer = engine->network()->addFill(nvinfer1::Dims{1, {1}}, - nvinfer1::FillOperation::kLINSPACE); - fill_layer->setInput(0, *shape_tensor); // 设置output shape - - at::Tensor value_tensor = torch::tensor(0).to(input_type); - nvinfer1::ITensor* value_itensor = tensor_to_const(engine, value_tensor); - fill_layer->setInput(1, *value_itensor); // 初始值 - - at::Tensor delta_tensor = torch::zeros(self_rank).to(input_type); // 每个方向上的变化,所以self_rank个0 - nvinfer1::ITensor* delta_itensor = tensor_to_const(engine, delta_tensor); - fill_layer->setInput(2, *delta_itensor); - fill_layer->setName((layer_info(node) + "_IFillLayer").c_str()); - - nvinfer1::ITensor* output = fill_layer->getOutput(0); - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - return true; - } -} - -// aten::zeros(int[] size, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None) -> Tensor" -bool ZerosConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 5), "invaid inputs size for ZerosConverter"); - - bool has_tensor_scalar = false; - has_tensor_scalar = check_inputs_tensor_scalar(engine, node); - nvinfer1::ITensor* output = nullptr; - // extract dtype - torch::jit::IValue maybe_type = engine->context().get_constant(inputs[1]); - - if (has_tensor_scalar) { - // extract size - nvinfer1::ITensor* shape_tensor = engine->context().get_tensor(inputs[0]); // from size - POROS_CHECK_TRUE((shape_tensor != nullptr), "Unable to init input tensor for node: " << *node); - - nvinfer1::Dims self_dims = shape_tensor->getDimensions(); - int64_t self_rank = self_dims.d[0]; - - nvinfer1::IFillLayer* fill_layer = engine->network()->addFill(nvinfer1::Dims{1, {1}}, - nvinfer1::FillOperation::kLINSPACE); - fill_layer->setInput(0, *shape_tensor); // 设置output shape - // default type is float - at::Tensor value_tensor = torch::tensor(0.0, torch::kFloat32); - at::Tensor delta_tensor = torch::zeros(self_rank, torch::kFloat32); // 每个方向上的变化,所以self_rank个0 - // type conversion - if (!maybe_type.isNone()) { - value_tensor = value_tensor.to(maybe_type.toScalarType()); - delta_tensor = delta_tensor.to(maybe_type.toScalarType()); - } - nvinfer1::ITensor* value_itensor = tensor_to_const(engine, value_tensor); - fill_layer->setInput(1, *value_itensor); // 初始值 - nvinfer1::ITensor* delta_itensor = tensor_to_const(engine, delta_tensor); - fill_layer->setInput(2, *delta_itensor); - fill_layer->setName((layer_info(node) + "_IFillLayer").c_str()); - output = fill_layer->getOutput(0); - } else { - std::vector self_vec = (engine->context().get_constant(inputs[0])).toIntList().vec(); - at::Tensor value_tensor = torch::zeros(self_vec, torch::kFloat32); - if (!maybe_type.isNone()) { - value_tensor = value_tensor.to(maybe_type.toScalarType()); - } - output = tensor_to_const(engine, value_tensor); - } - - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - return true; -} - - -// aten::ones(int[] size, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None) -> Tensor -bool OnesConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 5), "invaid inputs size for OnesConverter"); - - bool has_tensor_scalar = false; - has_tensor_scalar = check_inputs_tensor_scalar(engine, node); - nvinfer1::ITensor* output = nullptr; - // extract dtype - torch::jit::IValue maybe_type = engine->context().get_constant(inputs[1]); - - if (has_tensor_scalar) { - // extract size - nvinfer1::ITensor* shape_tensor = engine->context().get_tensor(inputs[0]); // from size - POROS_CHECK_TRUE((shape_tensor != nullptr), "Unable to init input tensor for node: " << *node); - - nvinfer1::Dims self_dims = shape_tensor->getDimensions(); - int64_t self_rank = self_dims.d[0]; - - nvinfer1::IFillLayer* fill_layer = engine->network()->addFill(nvinfer1::Dims{1, {1}}, - nvinfer1::FillOperation::kLINSPACE); - fill_layer->setInput(0, *shape_tensor); // 设置output shape - // default type is float - at::Tensor value_tensor = torch::tensor(1.0, torch::kFloat32); - at::Tensor delta_tensor = torch::zeros(self_rank, torch::kFloat32); // 每个方向上的变化,所以self_rank个0 - // type conversion - if (!maybe_type.isNone()) { - value_tensor = value_tensor.to(maybe_type.toScalarType()); - delta_tensor = delta_tensor.to(maybe_type.toScalarType()); - } - nvinfer1::ITensor* value_itensor = tensor_to_const(engine, value_tensor); - fill_layer->setInput(1, *value_itensor); // 初始值 - nvinfer1::ITensor* delta_itensor = tensor_to_const(engine, delta_tensor); - fill_layer->setInput(2, *delta_itensor); - fill_layer->setName((layer_info(node) + "_IFillLayer").c_str()); - output = fill_layer->getOutput(0); - } else { - std::vector self_vec = (engine->context().get_constant(inputs[0])).toIntList().vec(); - at::Tensor value_tensor = torch::ones(self_vec, torch::kFloat32); - if (!maybe_type.isNone()) { - value_tensor = value_tensor.to(maybe_type.toScalarType()); - } - output = tensor_to_const(engine, value_tensor); - } - - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - return true; -} - - -// aten::full(int[] size, Scalar fill_value, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None) -> Tensor -bool FullConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 6), "invaid inputs size for FullConverter"); - - bool has_tensor_scalar = false; - has_tensor_scalar = check_inputs_tensor_scalar(engine, node); - nvinfer1::ITensor* output = nullptr; - // extract fill_value - torch::jit::IValue maybe_value = engine->context().get_constant(inputs[1]); - POROS_CHECK_TRUE((!maybe_value.isNone()), "Unable to init input fill value for node: " << *node); - float fill_value = maybe_value.toScalar().toFloat(); - // extract dtype - torch::jit::IValue maybe_type = engine->context().get_constant(inputs[2]); - - if (has_tensor_scalar) { - // extract size - nvinfer1::ITensor* shape_tensor = engine->context().get_tensor(inputs[0]); // from size - POROS_CHECK_TRUE((shape_tensor != nullptr), "Unable to init input tensor for node: " << *node); - - nvinfer1::Dims self_dims = shape_tensor->getDimensions(); - int64_t self_rank = self_dims.d[0]; - - nvinfer1::IFillLayer* fill_layer = engine->network()->addFill(nvinfer1::Dims{1, {1}}, - nvinfer1::FillOperation::kLINSPACE); - fill_layer->setInput(0, *shape_tensor); // 设置output shape - // default type is float - at::Tensor value_tensor = torch::tensor(fill_value, torch::kFloat32); - at::Tensor delta_tensor = torch::zeros(self_rank, torch::kFloat32); // 每个方向上的变化,所以self_rank个0 - // type conversion - if (!maybe_type.isNone()) { - value_tensor = value_tensor.to(maybe_type.toScalarType()); - delta_tensor = delta_tensor.to(maybe_type.toScalarType()); - } - nvinfer1::ITensor* value_itensor = tensor_to_const(engine, value_tensor); - fill_layer->setInput(1, *value_itensor); // 初始值 - nvinfer1::ITensor* delta_itensor = tensor_to_const(engine, delta_tensor); - fill_layer->setInput(2, *delta_itensor); - fill_layer->setName((layer_info(node) + "_IFillLayer").c_str()); - output = fill_layer->getOutput(0); - } else { - std::vector self_vec = (engine->context().get_constant(inputs[0])).toIntList().vec(); - at::Tensor value_tensor = torch::ones(self_vec, torch::kFloat32) * fill_value; - if (!maybe_type.isNone()) { - value_tensor = value_tensor.to(maybe_type.toScalarType()); - } - output = tensor_to_const(engine, value_tensor); - } - - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - return true; -} - -// reduce input_tensor with shape -static nvinfer1::ITensor* reduce_dim1_to_dim0(TensorrtEngine* engine, nvinfer1::ITensor* input_tensor) { - nvinfer1::Dims input_dims = input_tensor->getDimensions(); - if (input_dims.nbDims == 1 && input_dims.d[0] == 1) { - nvinfer1::IShuffleLayer* shuffle_l = engine->network()->addShuffle(*input_tensor); - nvinfer1::Dims squeeze_dim; - squeeze_dim.nbDims = 0; - shuffle_l->setReshapeDimensions(squeeze_dim); - return shuffle_l->getOutput(0); - } else { - return input_tensor; - } -} - -// aten::arange(Scalar end, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None) -> Tensor -// aten::arange.start(Scalar start, Scalar end, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None) -> (Tensor) -bool ArangeConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 5 || inputs.size() == 6), "invaid inputs size for ArangeConverter."); - // start、end目前只支持int类型 - POROS_CHECK_TRUE((node->inputs()[0]->type()->kind() == c10::TypeKind::IntType), - "The type of input[0] for ArangeConverter must be Int."); - if (inputs.size() == 6) { - POROS_CHECK_TRUE((node->inputs()[1]->type()->kind() == c10::TypeKind::IntType), - "The type of input[1] for ArangeConverter must be Int."); - } - - int type_input_index = inputs.size() - 4; - torch::jit::IValue maybe_type = engine->context().get_constant(inputs[type_input_index]); - - if (check_inputs_tensor_scalar(engine, node)) { - nvinfer1::IFillLayer* fill_layer = engine->network()->addFill(nvinfer1::Dims{1, {1}}, - nvinfer1::FillOperation::kLINSPACE); - if (inputs.size() == 5) { - nvinfer1::ITensor* end_tensor = this->get_tensor_scalar(inputs[0]); - // 设置output shape - fill_layer->setInput(0, *end_tensor); - // 设置 start 和 delta - at::Tensor value_tensor = torch::tensor(0, torch::kInt32); - at::Tensor delta_tensor = torch::ones(1, torch::kInt32); - auto value_itensor = tensor_to_const(engine, value_tensor); - fill_layer->setInput(1, *value_itensor); - auto delta_itensor = tensor_to_const(engine, delta_tensor); - fill_layer->setInput(2, *delta_itensor); - } else { - nvinfer1::ITensor* start_tensor = this->get_tensor_scalar(inputs[0]); - nvinfer1::ITensor* end_tensor = this->get_tensor_scalar(inputs[1]); - // arange_size = end - start - nvinfer1::ITensor* arange_size = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUB, - end_tensor, - start_tensor, - layer_info(node) + "_get_arange_size")->getOutput(0); - // 设置output shape - fill_layer->setInput(0, *arange_size); - // 设置 start 和 delta - start_tensor = reduce_dim1_to_dim0(engine, start_tensor); - fill_layer->setInput(1, *start_tensor); - at::Tensor delta_tensor = torch::ones(1, torch::kInt32); - auto delta_itensor = tensor_to_const(engine, delta_tensor); - fill_layer->setInput(2, *delta_itensor); - } - fill_layer->setName((layer_info(node) + "_IFillLayer").c_str()); - nvinfer1::ITensor* output = fill_layer->getOutput(0); - - if (!maybe_type.isNone()) { - at::ScalarType scalar_type = maybe_type.toScalarType(); - if (scalar_type == at::ScalarType::Long) { - scalar_type = at::ScalarType::Int; - LOG(WARNING) << "aten::arange Converter meets c10::ScalarType::Long tensor type, change this to c10::ScalarType::Int. " - << "Attention: this may leed to percision change"; - } - nvinfer1::DataType output_type = attype_to_nvtype(scalar_type); - // Set datatype for data to dtype - auto identity = engine->network()->addIdentity(*output); - identity->setName((layer_info(node) + "_identity_output").c_str()); - identity->setOutputType(0, output_type); - output = identity->getOutput(0); - } - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - - } else { - at::Tensor value_tensor; - if (inputs.size() == 5) { - int64_t end = engine->context().get_constant(inputs[0]).toInt(); - value_tensor = torch::arange(end, torch::kInt); - - } else { - int64_t start = engine->context().get_constant(inputs[0]).toInt(); - int64_t end = engine->context().get_constant(inputs[1]).toInt(); - value_tensor = torch::arange(start, end, torch::kInt); - } - if (!maybe_type.isNone()) { - value_tensor = value_tensor.to(maybe_type.toScalarType()); - } - nvinfer1::ITensor* output = tensor_to_const(engine, value_tensor); - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - } - return true; -} - -// aten::tensor(t[] data, *, int? dtype=None, Device? device=None, bool requires_grad=False) -> (Tensor) -bool TensorConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 4), "invaid inputs size for TensorConverter"); - // extract dtype - torch::jit::IValue maybe_type = engine->context().get_constant(inputs[1]); - - nvinfer1::ITensor* output = nullptr; - if (check_inputs_tensor_scalar(engine, node)) { - output = this->get_tensor_scalar(inputs[0]); - if (!maybe_type.isNone()) { - at::ScalarType scalar_type = maybe_type.toScalarType(); - if (scalar_type == at::ScalarType::Long) { - scalar_type = at::ScalarType::Int; - LOG(WARNING) << "aten::tensor Converter meets c10::ScalarType::Long tensor type, change this to c10::ScalarType::Int. " - << "Attention: this may leed to percision change"; - } - auto output_type = attype_to_nvtype(scalar_type); - // Set datatype for data to dtype - auto identity = engine->network()->addIdentity(*output); - identity->setName((layer_info(node) + "_IIdentityLayer_for_output").c_str()); - identity->setOutputType(0, output_type); - output = identity->getOutput(0); - } - // mark: 06.30 by tsq - // 如果schema为aten::tensor.int,当其输出为子图输出时,即它的output会输出到torchscript,那么需要变换output rank为0。否则会出core。 - // 理论上来说应该所有aten::tensor.int输出的tensor rank都为0,但这样输出output->getDimensions()为空[] - // 暂时不清楚rank为0的nvtensor给其他op会有什么影响,所以先限制aten::tensor输出为子图输出时才squeeze - bool need_squeeze_dim = false; - if (node->hasUses()) { - auto users_list = node->output(0)->uses(); - for (size_t i = 0; i < users_list.size(); i++) { - if (users_list[i].user->kind() == torch::jit::prim::Return) { - need_squeeze_dim = true; - break; - } - } - } - - if (need_squeeze_dim && inputs[0]->type()->kind() == c10::TypeKind::IntType) { - nvinfer1::IShuffleLayer* shuffle_l = engine->network()->addShuffle(*output); - nvinfer1::Dims squeeze_dim; - squeeze_dim.nbDims = 0; - shuffle_l->setReshapeDimensions(squeeze_dim); - shuffle_l->setName((layer_info(node) + "_IShuffleLayer").c_str()); - output = shuffle_l->getOutput(0); - engine->context().set_tensor(node->outputs()[0], output); - return true; - } - - } else { - // extract dtype - at::Tensor input_data = engine->context().get_constant(inputs[0]).toTensor(); - if (!maybe_type.isNone()) { - input_data = input_data.to(maybe_type.toScalarType()); - } - output = tensor_to_const(engine, input_data); - } - - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - return true; -} - -// aten::linspace(Scalar start, Scalar end, int? steps=None, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None) -> (Tensor) -bool LinspaceConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 7), "invaid inputs size for LinspaceConverter."); - - bool has_tensor_scalar = check_inputs_tensor_scalar(engine, node); - - auto fill_layer = engine->network()->addFill(nvinfer1::Dims{1, {1}}, nvinfer1::FillOperation::kLINSPACE); - if (has_tensor_scalar) { - nvinfer1::ITensor* start = this->get_tensor_scalar(inputs[0]); - nvinfer1::ITensor* end = this->get_tensor_scalar(inputs[1]); - nvinfer1::ITensor* step = this->get_tensor_scalar(inputs[2]); - // steps为None时默认值为100 - if (step == nullptr) { - step = tensor_to_const(engine, at::tensor({100}, at::ScalarType::Float)); - } - // 默认输出类型为float,以下操作也需要转为float - // alpha=start,delta=(end - start) / (step - 1) - if (start->getType() != nvinfer1::DataType::kFLOAT) { - auto identity = engine->network()->addIdentity(*start); - identity->setOutputType(0, nvinfer1::DataType::kFLOAT); - identity->setName((layer_info(node) + "_IIdentityLayer_for_start").c_str()); - start = identity->getOutput(0); - } - if (end->getType() != nvinfer1::DataType::kFLOAT) { - auto identity = engine->network()->addIdentity(*end); - identity->setOutputType(0, nvinfer1::DataType::kFLOAT); - identity->setName((layer_info(node) + "_IIdentityLayer_for_end").c_str()); - end = identity->getOutput(0); - } - if (step->getType() != nvinfer1::DataType::kFLOAT) { - auto identity = engine->network()->addIdentity(*step); - identity->setOutputType(0, nvinfer1::DataType::kFLOAT); - identity->setName((layer_info(node) + "_IIdentityLayer_for_step").c_str()); - step = identity->getOutput(0); - } - // (end - start) - nvinfer1::ILayer* sub_layer = add_elementwise(engine, nvinfer1::ElementWiseOperation::kSUB, - end, start, layer_info(node) + "_sub(end_start)"); - nvinfer1::ITensor* length = sub_layer->getOutput(0); - // (step - 1) - nvinfer1::ITensor* one = tensor_to_const(engine, at::tensor({1}, at::ScalarType::Float)); - nvinfer1::ILayer* sub_layer2 = add_elementwise(engine, nvinfer1::ElementWiseOperation::kSUB, - step, one, layer_info(node) + "_sub(step_one)"); - nvinfer1::ITensor* step_sub_one = sub_layer2->getOutput(0); - // (end - start) / (step - 1) - nvinfer1::ILayer* div_layer = add_elementwise(engine, nvinfer1::ElementWiseOperation::kDIV, - length, step_sub_one, layer_info(node) + "_div(get_delta)"); - nvinfer1::ITensor* delta = div_layer->getOutput(0); - // step需要转回int32作为Ifilllayer input0的输入,用于指定输出的dim - if (step->getType() == nvinfer1::DataType::kFLOAT) { - auto identity = engine->network()->addIdentity(*step); - identity->setOutputType(0, nvinfer1::DataType::kINT32); - identity->setName((layer_info(node) + "_IIdentityLayer_for_step_back").c_str()); - step = identity->getOutput(0); - } - // 输出只有一维,Ifilllayer需要start的rank为0(check_inputs_tensor_scalar中start scalar转nvtensor时自带了1维) - if (start->getDimensions().nbDims > 0) { - nvinfer1::IShuffleLayer* shuffle_l = engine->network()->addShuffle(*start); - nvinfer1::Dims start_dim; - start_dim.nbDims = 0; - shuffle_l->setReshapeDimensions(start_dim); - shuffle_l->setName((layer_info(node) + "_IShuffleLayer_for_start").c_str()); - start = shuffle_l->getOutput(0); - } - fill_layer->setInput(0, *step); - fill_layer->setInput(1, *start); - fill_layer->setInput(2, *delta); - } else { - torch::jit::IValue start_ivalue = engine->context().get_constant(inputs[0]); - torch::jit::IValue end_ivalue = engine->context().get_constant(inputs[1]); - torch::jit::IValue maybe_step = engine->context().get_constant(inputs[2]); - float start = start_ivalue.toScalar().to(); - float end = end_ivalue.toScalar().to(); - float step = 100.0; - if (!maybe_step.isNone()) { - step = maybe_step.toScalar().to(); - } - float delta = (end - start) / (step - 1); - std::vector output_dims = {(int64_t)step}; - fill_layer->setDimensions(sizes_to_nvdim(output_dims)); - fill_layer->setAlpha(start); - fill_layer->setBeta(delta); - } - - fill_layer->setName((layer_info(node) + "_IFillLayer").c_str()); - nvinfer1::ITensor* output = fill_layer->getOutput(0); - - // extract dtype - torch::jit::IValue maybe_type = engine->context().get_constant(inputs[3]); - // 如果输出不为空,则最后变换输出类型 - if (!maybe_type.isNone()) { - nvinfer1::DataType output_type = attype_to_nvtype(maybe_type.toScalarType()); - auto identity = engine->network()->addIdentity(*output); - identity->setName((layer_info(node) + "_IIdentityLayer_for_output").c_str()); - identity->setOutputType(0, output_type); - output = identity->getOutput(0); - } - - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - return true; -} - -// aten::full_like(Tensor self, Scalar fill_value, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None, int? memory_format=None) -> (Tensor) -bool FulllikeConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 7), "invaid inputs size for FulllikeConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for FulllikeConverter is not Tensor as expected"); - // extract self - nvinfer1::ITensor* self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - nvinfer1::Dims self_dims = self->getDimensions(); - - // 先转成float去接input[1]输入 - auto scalar_ivalue = (engine->context().get_constant(inputs[1])); - float scalar = scalar_ivalue.toScalar().to(); - - // extract type - torch::jit::IValue maybe_type = engine->context().get_constant(inputs[2]); - - bool is_dynamic = check_nvtensor_is_dynamic(self); - - nvinfer1::IFillLayer* fill_layer = engine->network()->addFill(nvinfer1::Dims{1, {1}}, - nvinfer1::FillOperation::kLINSPACE); - // set fill shape - if (is_dynamic) { - nvinfer1::ITensor* shape_tensor = engine->network()->addShape(*self)->getOutput(0); - fill_layer->setInput(0, *shape_tensor); - } else { - fill_layer->setDimensions(self_dims); - } - - at::ScalarType init_type = (inputs[0]->type()->cast())->scalarType().value(); - if (init_type == at::ScalarType::Long) { - init_type = at::ScalarType::Int; - } else if (init_type == at::ScalarType::Double) { - init_type = at::ScalarType::Float; - } - // 默认输出类型和self一致,与torch保持一致 - at::Tensor value_tensor = torch::tensor(scalar, {init_type}); - at::Tensor delta_tensor = torch::zeros(self_dims.nbDims, {init_type}); // 每个方向上的变化,所以self_rank个0 - if (!maybe_type.isNone()) { - at::ScalarType input_type = maybe_type.toScalarType(); - if (input_type == at::ScalarType::Long) { - input_type = at::ScalarType::Int; - } else if (input_type == at::ScalarType::Double) { - input_type = at::ScalarType::Float; - } - value_tensor = value_tensor.to(input_type); - delta_tensor = delta_tensor.to(input_type); - } - - nvinfer1::ITensor* value_itensor = tensor_to_const(engine, value_tensor); - fill_layer->setInput(1, *value_itensor); // 初始值 - - nvinfer1::ITensor* delta_itensor = tensor_to_const(engine, delta_tensor); - fill_layer->setInput(2, *delta_itensor); - nvinfer1::ITensor* output = fill_layer->getOutput(0); - - fill_layer->setName((layer_info(node) + "_IFillLayer").c_str()); - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, ZerosLikeConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, ZerosConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, OnesConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, FullConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, ArangeConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, TensorConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, LinspaceConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, FulllikeConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/generate.h b/poros/poros/converter/gpu/generate.h deleted file mode 100644 index 22775480dfe..00000000000 --- a/poros/poros/converter/gpu/generate.h +++ /dev/null @@ -1,212 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file generate.h -* @author tianshaoqing@baidu.com -* @date Mon Dec 6 14:29:20 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include -#include - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -// Tensor zeros_like(const Tensor & self, c10::optional dtype, c10::optional layout, c10::optional device, c10::optional pin_memory, c10::optional memory_format); -class ZerosLikeConverter : public GpuConverter { -public: - ZerosLikeConverter() {} - virtual ~ZerosLikeConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::zeros_like(Tensor self, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None, MemoryFormat? memory_format=None) -> Tensor"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::zeros_like}; - } -}; - -// Tensor zeros(IntArrayRef size, c10::optional dtype, c10::optional layout, c10::optional device, c10::optional pin_memory); -// aten::zeros(int[] size, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None) -> Tensor" -class ZerosConverter : public GpuConverter { -public: - ZerosConverter() {} - virtual ~ZerosConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::zeros(int[] size, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None) -> Tensor"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::zeros}; - } - - bool assign_schema_attr() { - return assign_schema_attr_helper({{"aten::zeros(int[] size, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None) -> Tensor", {1, 1}}}); - } -}; - -class OnesConverter : public GpuConverter { -public: - OnesConverter() {} - virtual ~OnesConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::ones(int[] size, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None) -> Tensor"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::ones}; - } - - bool assign_schema_attr() { - return assign_schema_attr_helper({{"aten::ones(int[] size, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None) -> Tensor", {1, 1}}}); - } -}; - -class FullConverter : public GpuConverter { -public: - FullConverter() {} - virtual ~FullConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::full(int[] size, Scalar fill_value, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None) -> Tensor"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::full}; - } - - bool assign_schema_attr() { - return assign_schema_attr_helper({{"aten::full(int[] size, Scalar fill_value, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None) -> Tensor", {1, 1}}}); - } -}; - -class ArangeConverter : public GpuConverter { -public: - ArangeConverter() {} - virtual ~ArangeConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::arange(Scalar end, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None) -> Tensor", - "aten::arange.start(Scalar start, Scalar end, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None) -> (Tensor)",}; - } - - const std::vector node_kind() { - return {torch::jit::aten::arange}; - } - - bool assign_schema_attr() { - bool result = true; - result &= assign_schema_attr_helper({{"aten::arange(Scalar end, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None) -> Tensor", {1, 1}}}); - result &= assign_schema_attr_helper({{"aten::arange.start(Scalar start, Scalar end, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None) -> (Tensor)", {1, 1}}}); - return result; - } -}; - -class TensorConverter : public GpuConverter { -public: - TensorConverter() {} - virtual ~TensorConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::tensor(t[] data, *, int? dtype=None, Device? device=None, bool requires_grad=False) -> (Tensor)", - "aten::tensor.int(int t, *, int? dtype=None, Device? device=None, bool requires_grad=False) -> (Tensor)"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::tensor}; - } - - bool assign_schema_attr() { - bool result = true; - result &= assign_schema_attr_helper({{"aten::tensor(t[] data, *, int? dtype=None, Device? device=None, bool requires_grad=False) -> (Tensor)", {1, 1}}}); - result &= assign_schema_attr_helper({{"aten::tensor.int(int t, *, int? dtype=None, Device? device=None, bool requires_grad=False) -> (Tensor)", {1, 1}}}); - return result; - } -}; - -class LinspaceConverter : public GpuConverter { -public: - LinspaceConverter() {} - virtual ~LinspaceConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector node_kind() { - return {torch::jit::aten::linspace}; - } - - // aten::linspace schema changed in torch-1.11 - const std::vector schema_string() { - if (TORCH_VERSION_MAJOR < 2 && TORCH_VERSION_MINOR < 11) { - return {"aten::linspace(Scalar start, Scalar end, int? steps=None, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None) -> (Tensor)",}; - } else { - return {"aten::linspace(Scalar start, Scalar end, int steps, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None) -> (Tensor)",}; - } - } - - // aten::linspace schema changed in torch-1.11 - bool assign_schema_attr() { - if (TORCH_VERSION_MAJOR < 2 && TORCH_VERSION_MINOR < 11) { - return assign_schema_attr_helper({{"aten::linspace(Scalar start, Scalar end, int? steps=None, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None) -> (Tensor)", {1, 1}}}); - } else { - return assign_schema_attr_helper({{"aten::linspace(Scalar start, Scalar end, int steps, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None) -> (Tensor)", {1, 1}}}); - } - } -}; - -class FulllikeConverter : public GpuConverter { -public: - FulllikeConverter() {} - virtual ~FulllikeConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::full_like(Tensor self, Scalar fill_value, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None, int? memory_format=None) -> (Tensor)",}; - } - - const std::vector node_kind() { - return {torch::jit::aten::full_like}; - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/converter/gpu/gpu_converter.cpp b/poros/poros/converter/gpu/gpu_converter.cpp deleted file mode 100644 index 4d23e77b0a2..00000000000 --- a/poros/poros/converter/gpu/gpu_converter.cpp +++ /dev/null @@ -1,100 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file gpu_converter.cpp -* @author tianshaoqing@baidu.com -* @date Mon Dec 27 11:24:21 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/gpu_converter.h" - -#include "poros/converter/gpu/weight.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -bool GpuConverter::check_inputs_tensor_scalar(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - bool has_tensor_scalar = false; - // 检查int或int[]类型的输入是否包含nvtensor - for (size_t i = 0; i < inputs.size(); i++) { - const torch::jit::Value* node_input = inputs[i]; - // int, int[] or float32 - if (node_input->type()->kind() == c10::TypeKind::IntType || - node_input->type()->isSubtypeOf(c10::ListType::ofInts()) || - node_input->type()->str() == "float") { - if (engine->context().get_tensor(node_input) != nullptr) { - // 检查成功立刻停止循环 - LOG(INFO) << node_info(node) << ": inputs[" << i << "] is tensor scalar."; - has_tensor_scalar = true; - break; - } - } - } - // 如果int或int[]中包含nvtensor, 将所有输入int与nvtensor建立映射到map中 - if (has_tensor_scalar) { - _tensor_scalar_map.clear(); - for (size_t i = 0; i < inputs.size(); i++) { - const torch::jit::Value* node_input = inputs[i]; - // int or int[] - // 2022.5.13 @wangrui39: 这里加了float,这个情况会用在aten::div(Scalar a, Scalar b) -> (float) - if (node_input->type()->kind() == c10::TypeKind::IntType || - node_input->type()->isSubtypeOf(c10::ListType::ofInts()) || - node_input->type()->str() == "float") { - nvinfer1::ITensor* temp = engine->context().get_tensor(inputs[i]); - // 若直接获取到了nvtensor, 直接建立映射 - if (temp != nullptr) { - _tensor_scalar_map.emplace(inputs[i], temp); - } else { - // 若未获取int或int[]对应到nvtensor, get其ivalue值再转成nvtensor, 建立映射关系 - torch::jit::IValue temp_ivalue = engine->context().get_constant(inputs[i]); - if (temp_ivalue.isInt()) { - int64_t temp_int = temp_ivalue.toScalar().to(); - _tensor_scalar_map.emplace(inputs[i], - tensor_to_const(engine, torch::tensor({temp_int}, torch::kInt))); - } else if (temp_ivalue.type()->str() == "float") { - float temp_float = temp_ivalue.toScalar().to(); - _tensor_scalar_map.emplace(inputs[i], - tensor_to_const(engine, torch::tensor({temp_float}, torch::kFloat))); - } else if (temp_ivalue.isIntList()){ - _tensor_scalar_map.emplace(inputs[i], - tensor_to_const(engine, torch::tensor(temp_ivalue.toIntList().vec(), torch::kInt))); - } else { - // 若获取ivalue也失败, 则建立int与空指针的关系, 外部获取后需判断 - _tensor_scalar_map.emplace(inputs[i], nullptr); - LOG(FATAL) << node_info(node) + std::string(" input[") + - std::to_string(i) + std::string("] get int ivalue false."); - } - } - } - } - } - return has_tensor_scalar; -} - -nvinfer1::ITensor* GpuConverter::get_tensor_scalar(const torch::jit::Value* value) { - auto it = _tensor_scalar_map.find(value); - if (it == _tensor_scalar_map.end()) { - return nullptr; - } - return it->second; -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/gpu_converter.h b/poros/poros/converter/gpu/gpu_converter.h deleted file mode 100644 index 32a3c967008..00000000000 --- a/poros/poros/converter/gpu/gpu_converter.h +++ /dev/null @@ -1,74 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file gpu_converter.h -* @author tianjinjin@baidu.com -* @author huangben@baidu.com -* @date Tue Jul 27 11:24:21 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/iconverter.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/log/poros_logging.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class GpuConverter : public IConverter { -public: - virtual ~GpuConverter() {} - virtual bool converter(TensorrtEngine* engine, const torch::jit::Node *node) = 0; - virtual bool converter(IEngine* engine, const torch::jit::Node *node) { - return converter(static_cast(engine), node); - } - virtual const std::vector schema_string() = 0; - virtual const std::vector node_kind() = 0; - -protected: - /** - * @brief Check whether the scalar inputs of the node are nvinfer1::ITensor (not come from prim::Constant). - * If yes, convert other scalar inputs to nvinfer1::ITensor and save them in _tensor_scalar_map. - * The type of nvinfer1::ITensor is consistent with the original scalar. - * - * @param [in] engine : TensorrtEngine - * @param [in] node : node in torch::jit::Graph - * @return bool - * @retval true => yes false => no - **/ - bool check_inputs_tensor_scalar(TensorrtEngine* engine, const torch::jit::Node *node); - /** - * @brief get nvinfer1::ITensor* type scalar from _tensor_scalar_map. - * - * @param [in] value : the input value of the node. - * @return nvinfer1::ITensor* - **/ - nvinfer1::ITensor* get_tensor_scalar(const torch::jit::Value* value); - -private: - std::unordered_map _tensor_scalar_map; -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/group_norm.cpp b/poros/poros/converter/gpu/group_norm.cpp deleted file mode 100644 index 7282c082f16..00000000000 --- a/poros/poros/converter/gpu/group_norm.cpp +++ /dev/null @@ -1,461 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file group_norm.cpp -* @author tianshaoqing@baidu.com -* @date Fri Jan 21 15:28:37 CST 2022 -* @brief -**/ - -#include "poros/converter/gpu/group_norm.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/converter/gpu/weight.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/** - * @brief expand_gamma_beta - * such as: - * input shape is [2, 10, 3, 3]. - * This function first shuffle gamma(or beta) shape from [10] to [10, 1, 1], - * then slice gamma(or beta) shape from [10, 1, 1] to [10, 3, 3]. - * - * @param [in] engine : engine of group_norm converter. - * @param [in] weight_tensor : gamma or beta tensor. - * @param [in] weight_shuffle_dims : gamma or beta shuffle dims. - * @param [in] target_size : if input is dynamic, this parameter determines the slice size. - * @param [in] target_dims : if input is not dynamic, this parameter determines the slice size. - * @param [in] is_dynamic : input is dynamic or not. - * @return nvinfer1::ITensor* - * @retval -**/ -static nvinfer1::ITensor* expand_gamma_beta(TensorrtEngine* engine, - nvinfer1::ITensor* weight_tensor, - const nvinfer1::Dims& weight_shuffle_dims, - nvinfer1::ITensor* target_size, - const nvinfer1::Dims& target_dims, - const bool& is_dynamic, - const std::string& name) { - nvinfer1::IShuffleLayer* shuffle_l = engine->network()->addShuffle(*weight_tensor); - shuffle_l->setReshapeDimensions(weight_shuffle_dims); - std::vector start(target_dims.nbDims, 0), stride(target_dims.nbDims, 0); - stride[0] = 1; - nvinfer1::ISliceLayer* slice_l = engine->network()->addSlice(*(shuffle_l->getOutput(0)), - sizes_to_nvdim(start), - target_dims, - sizes_to_nvdim(stride)); - if (is_dynamic) { - slice_l->setInput(2, *target_size); - } - slice_l->setName(name.c_str()); - return slice_l->getOutput(0); -} - -// aten::group_norm(Tensor input, int num_groups, Tensor? weight=None, Tensor? bias=None, float eps=1e-05, bool cudnn_enabled=True) -> Tensor -bool GroupNormConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 6), "invaid inputs size for GroupNormConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for GroupNormConverter is not Tensor as expected"); - // weight & bias - POROS_CHECK_TRUE((inputs[2]->node()->kind() == torch::jit::prim::Constant), - "input[2] for GroupNormConverter is not come from prim::Constant as expected"); - POROS_CHECK_TRUE((inputs[3]->node()->kind() == torch::jit::prim::Constant), - "input[3] for GroupNormConverter is not come from prim::Constant as expected"); - - nvinfer1::ITensor* input = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((input != nullptr), "Unable to init input tensor for node: " << *node); - - //extract rank of input and check it - int input_rank = input->getDimensions().nbDims; - if (input_rank < 2) { - LOG(WARNING) << *node << ": num of input dimensions must be greater than 2, but got " << input_rank; - return false; - } - - //extract tensor type info of input - at::ScalarType tensor_type = nvtype_to_attype(input->getType()); - auto options = torch::TensorOptions().dtype(tensor_type); - - //extract shape of input - nvinfer1::Dims ori_shape = input->getDimensions(); - nvinfer1::ITensor* ori_shape_tensor = engine->network()->addShape(*input)->getOutput(0); - std::vector ori_shape_vec = nvdim_to_sizes(ori_shape); - int64_t channel_size = ori_shape_vec[1]; - - //extract num_groups info - int64_t num_groups = (engine->context().get_constant(inputs[1])).toInt(); - - // check input is dynamic or not - bool is_dynamic = check_nvtensor_is_dynamic(input); - - if (!is_dynamic && channel_size % num_groups != 0) { - LOG(WARNING) << *node << ":Expected number of channels in input to be divisible by num_groups," - << " but got input of shape " << ori_shape << ", and num_groups=" << num_groups; - return false; - } - - // ATTENTION! we need to static_cast eps from double to float. otherwise coredump happened in instancenorm plugin - //double eps = (engine->context().get_constant(inputs[4])).toDouble(); - float eps = static_cast(engine->context().get_constant(inputs[4]).toDouble()); - - //reshape input - std::vector new_shape = {0, num_groups, -1}; - nvinfer1::ITensor* new_shape_tensor = tensor_to_const(engine, torch::tensor(new_shape, torch::kInt64)); - nvinfer1::IShuffleLayer* input_shuffle = engine->network()->addShuffle(*input); - input_shuffle->setInput(1, *new_shape_tensor); - input_shuffle->setName((layer_info(node) + "_IShuffleLayer_for_input").c_str()); - nvinfer1::ITensor* input_reshaped = input_shuffle->getOutput(0); - - // const std::vector expand_axes{3}; - // input_reshaped = unsqueeze_itensor(engine, input_reshaped, expand_axes); - nvinfer1::ITensor* norm_input = add_padding(engine, node, input_reshaped, 4); - - torch::Tensor weight_ = at::ones(num_groups, options).cpu().contiguous(); - torch::Tensor bias_ = at::zeros(num_groups, options).cpu().contiguous(); - - //set to instancenorm first - const int relu = 0; - const float alpha = 0; - std::vector f; - f.emplace_back(nvinfer1::PluginField("epsilon", &eps, nvinfer1::PluginFieldType::kFLOAT32, 1)); - f.emplace_back(nvinfer1::PluginField("scales", weight_.data_ptr(), nvinfer1::PluginFieldType::kFLOAT32, weight_.numel())); - f.emplace_back(nvinfer1::PluginField("bias", bias_.data_ptr(), nvinfer1::PluginFieldType::kFLOAT32, bias_.numel())); - f.emplace_back(nvinfer1::PluginField("relu", &relu, nvinfer1::PluginFieldType::kINT32, 1)); - f.emplace_back(nvinfer1::PluginField("alpha", &alpha, nvinfer1::PluginFieldType::kFLOAT32, 1)); - - nvinfer1::PluginFieldCollection fc; - fc.nbFields = f.size(); - fc.fields = f.data(); - - auto creator = getPluginRegistry()->getPluginCreator("InstanceNormalization_TRT", "1", ""); - auto instance_norm_plugin = creator->createPlugin("instance_norm", &fc); - - POROS_CHECK(instance_norm_plugin, "Unable to create instance_norm plugin from TensorRT plugin registry" << *node); - auto new_layer = engine->network()->addPluginV2( - reinterpret_cast(&norm_input), 1, *instance_norm_plugin); - new_layer->setName((layer_info(node) + "_plugin_instance_norm").c_str()); - nvinfer1::ITensor* norm_reshaped = new_layer->getOutput(0); - - nvinfer1::IShuffleLayer* norm_shuffle = engine->network()->addShuffle(*norm_reshaped); - norm_shuffle->setInput(1, *ori_shape_tensor); - norm_shuffle->setName((layer_info(node) + "_IShuffleLayer_for_input_back").c_str()); - nvinfer1::ITensor* norm = norm_shuffle->getOutput(0); - - std::vector axes(input_rank - 2); - std::iota(axes.begin(), axes.end(), 1); - - nvinfer1::ITensor* weight = engine->context().get_tensor(inputs[2]); - if (weight == nullptr) { - weight = tensor_to_const(engine, at::ones(1, options)); - } - weight = unsqueeze_itensor(engine, weight, axes); - - nvinfer1::ITensor* bias = engine->context().get_tensor(inputs[3]); - if (bias == nullptr) { - bias = tensor_to_const(engine, at::zeros(1, options)); - } - bias = unsqueeze_itensor(engine, bias, axes); - - //add(mul(norm, weight), bias) - nvinfer1::ITensor* mul_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - norm, - weight, - layer_info(node) + "_prod")->getOutput(0); - - nvinfer1::ITensor* final_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - mul_tensor, - bias, - layer_info(node) + "_sum")->getOutput(0); - - engine->context().set_tensor(node->outputs()[0], final_tensor); - LOG(INFO) << "Output tensor shape: " << final_tensor->getDimensions(); - return true; -} - -// aten::group_norm(Tensor input, int num_groups, Tensor? weight=None, Tensor? bias=None, float eps=1e-05, bool cudnn_enabled=True) -> Tensor -bool GroupNormConverter::converter_old(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 6), "invaid inputs size for GroupNormConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for GroupNormConverter is not Tensor as expected"); - // weight & bias - POROS_CHECK_TRUE((inputs[2]->node()->kind() == torch::jit::prim::Constant), - "input[2] for GroupNormConverter is not come from prim::Constant as expected"); - POROS_CHECK_TRUE((inputs[3]->node()->kind() == torch::jit::prim::Constant), - "input[3] for GroupNormConverter is not come from prim::Constant as expected"); - - nvinfer1::ITensor* input = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((input != nullptr), "Unable to init input tensor for node: " << *node); - nvinfer1::Dims ori_shape = input->getDimensions(); - - if (ori_shape.nbDims < 2) { - LOG(WARNING) << *node << ": num of input dimensions must be greater than 2, but got " << ori_shape.nbDims; - return false; - } - - std::vector ori_shape_vec = nvdim_to_sizes(ori_shape); - int64_t num_groups = (engine->context().get_constant(inputs[1])).toInt(); - - // check input is dynamic or not - bool is_dynamic = check_nvtensor_is_dynamic(input); - - if (!is_dynamic && ori_shape_vec[1] % num_groups != 0) { - LOG(WARNING) << *node << ":Expected number of channels in input to be divisible by num_groups," - << " but got input of shape " << ori_shape << ", and num_groups=" << num_groups; - return false; - } - - // Unwrap eps. - double eps = (engine->context().get_constant(inputs[4])).toDouble(); - std::vector input_groups; - std::vector output_groups; - - // divide input into num_group parts on channels - // such as: - // input shape is [2, 10, 3, 3] and num_group = 2. - // input is divided into 2 groups on channels, and each shape is [2, 5, 3, 3]. - std::vector start_vec(ori_shape_vec.size(), 0), size_vec(ori_shape_vec), stride_vec(ori_shape_vec.size(), 1); - std::vector group_channel_rev_mask_vec(ori_shape_vec.size(), 1); - std::vector group_channel_mask_vec(ori_shape_vec.size(), 0); - group_channel_rev_mask_vec[1] = 0; - group_channel_mask_vec[1] = 1; - nvinfer1::Dims start_dims, size_dims, stride_dims; - size_vec[1] = ori_shape_vec[1] / num_groups; - if (is_dynamic) { - for (size_t i = 0; i < size_vec.size(); i++) { - size_vec[i] = 0; - } - } - start_dims = sizes_to_nvdim(start_vec); - size_dims = sizes_to_nvdim(size_vec); - stride_dims = sizes_to_nvdim(stride_vec); - nvinfer1::ITensor* ori_shape_tensor = nullptr; - nvinfer1::ITensor* size_tensor = nullptr; - nvinfer1::ITensor* start_tensor = nullptr; - - if (is_dynamic) { - ori_shape_tensor = engine->network()->addShape(*input)->getOutput(0); - at::Tensor group_channel_rev_mask_tensor = torch::tensor(group_channel_rev_mask_vec, torch::kInt); - group_channel_rev_mask_tensor[1] = num_groups; - size_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kDIV, - ori_shape_tensor, - tensor_to_const(engine, group_channel_rev_mask_tensor), - (layer_info(node) + "_div_for_shape").c_str())->getOutput(0); - at::Tensor group_channel_mask_tensor = torch::tensor(group_channel_mask_vec, torch::kInt); - - start_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - size_tensor, - tensor_to_const(engine, group_channel_mask_tensor), - (layer_info(node) + "_prod_for_shape").c_str())->getOutput(0); - } - - for (int i = 0; i < num_groups; i++) { - start_dims.d[1] = size_vec[1] * i; - nvinfer1::ISliceLayer* slice_l = engine->network()->addSlice(*input, start_dims, size_dims, stride_dims); - if (is_dynamic) { - nvinfer1::ITensor* start_it_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - start_tensor, - tensor_to_const(engine, torch::tensor(i, torch::kInt)), - (layer_info(node) + "_prod_for_start_" + std::to_string(i)).c_str())->getOutput(0); - slice_l->setInput(1, *start_it_tensor); - slice_l->setInput(2, *size_tensor); - slice_l->setName((layer_info(node) + "_ISliceLayer_" + std::to_string(i)).c_str()); - } - input_groups.push_back(slice_l->getOutput(0)); - } - // calculate (x - E[x]) / sqrt((var + eps)) for each group - for (size_t i = 0; i < input_groups.size(); i++) { - // Set up axis_ask for E[x]. - uint32_t axis_mask = 0; - for (size_t i = 0; i < ori_shape_vec.size() - 1; i++) { - axis_mask |= 1 << (ori_shape_vec.size() - i - 1); - } - LOG(INFO) << "Axis Mask for E[x]" << std::bitset<32>(axis_mask); - - // E[x] - nvinfer1::IReduceLayer* mean_expected = engine->network()->addReduce(*input_groups[i], - nvinfer1::ReduceOperation::kAVG, axis_mask, true); - POROS_CHECK(mean_expected, "Unable to create mean_expected from node: " << *node); - mean_expected->setName((layer_info(node) + "_IReduceLayer(mean_expected)_" + std::to_string(i)).c_str()); - nvinfer1::ITensor* mean_expected_out = mean_expected->getOutput(0); - - // X-E[x] - nvinfer1::ILayer* sub = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUB, - input_groups[i], - mean_expected_out, - (layer_info(node) + "_sub_" + std::to_string(i)).c_str()); - POROS_CHECK(sub, "Unable to create Sub layer from node: " << *node); - nvinfer1::ITensor* xsubmean_out = sub->getOutput(0); - - // Variance = mean(pow(xsubmean,2)) - float pow_scalar = 2.0; - nvinfer1::ITensor* exponent = tensor_to_const(engine, torch::tensor({pow_scalar}, torch::kFloat)); - nvinfer1::ILayer* pow = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPOW, - xsubmean_out, - exponent, - (layer_info(node) + "_pow_" + std::to_string(i)).c_str()); - POROS_CHECK(pow, "Unable to create Pow layer from node: " << *node); - nvinfer1::ITensor* pow_out = pow->getOutput(0); - - nvinfer1::IReduceLayer* mean_var = engine->network()->addReduce(*pow_out, - nvinfer1::ReduceOperation::kAVG, axis_mask, true); - POROS_CHECK(mean_var, "Unable to create mean_var from node: " << *node); - mean_var->setName((layer_info(node) + "_IReduceLayer(mean_var)_" + std::to_string(i)).c_str()); - nvinfer1::ITensor* mean_var_out = mean_var->getOutput(0); - - // Variance + eps - nvinfer1::ITensor* eps_tensor = tensor_to_const(engine, torch::tensor({eps}, torch::kFloat)); - nvinfer1::ILayer* add = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - mean_var_out, - eps_tensor, - (layer_info(node) + "_add_" + std::to_string(i)).c_str()); - POROS_CHECK(add, "Unable to create Add layer from node: " << *node); - nvinfer1::ITensor* add_out = add->getOutput(0); - - // SQRT((Var + eps)) - nvinfer1::IUnaryLayer* sqrt = engine->network()->addUnary(*add_out, nvinfer1::UnaryOperation::kSQRT); - POROS_CHECK(sqrt, "Unable to create unary(sqrt) from node: " << *node); - sqrt->setName((layer_info(node) + "_IUnaryLayer_" + std::to_string(i)).c_str()); - nvinfer1::ITensor* sqrt_out = sqrt->getOutput(0); - - // (x - E[x]) / sqrt((var + eps)) - nvinfer1::ILayer* div = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kDIV, - xsubmean_out, - sqrt_out, - (layer_info(node) + "_div_" + std::to_string(i)).c_str()); - POROS_CHECK(div, "Unable to create div layer from node: " << *node); - nvinfer1::ITensor* div_out = div->getOutput(0); - output_groups.push_back(div_out); - } - nvinfer1::IConcatenationLayer* cat_layer = engine->network()->addConcatenation(output_groups.data(), - output_groups.size()); - cat_layer->setAxis(1); - cat_layer->setName((layer_info(node) + "_IConcatenationLayer").c_str()); - nvinfer1::ITensor* cat_out = cat_layer->getOutput(0); - engine->context().set_tensor(node->outputs()[0], cat_out); - - torch::jit::IValue maybe_weight = engine->context().get_constant(inputs[2]); - torch::jit::IValue maybe_bias = engine->context().get_constant(inputs[3]); - //when weight and bias setting is both None - if (!maybe_weight.isTensor() && !maybe_bias.isTensor()) { - engine->context().set_tensor(node->outputs()[0], cat_out); - LOG(INFO) << "Output tensor shape: " << cat_out->getDimensions(); - return true; - } - - /*------------------------------------------------------------ - * situation when weight or bias setting is not None - * ------------------------------------------------------------*/ - // Remove batch dimension from input shape for expand_size, which will - // be used to create weights for addScaleNd later. - - /** TODO: IS the first input size always are always be batch????? - * if not, this converter is not ok。 - * */ - - nvinfer1::ILayer* scale_l = nullptr; - nvinfer1::ILayer* shift_l = nullptr; - std::vector weights_dims_vec; - std::vector weights_shuffle_dims_vec(ori_shape_vec.size() - 1, 1); - weights_shuffle_dims_vec[0] = ori_shape_vec[1]; - weights_dims_vec.insert(weights_dims_vec.end(), ori_shape_vec.begin() + 1, ori_shape_vec.end()); - nvinfer1::ITensor* weights_shape_tensor = nullptr; - // although shape of input is dynamic, its rank is fix. so we can remove its batch dim to get expand size. - if (is_dynamic) { - for (size_t i = 0; i < weights_dims_vec.size(); i++) { - weights_dims_vec[i] = 0; - } - std::vector start = {1}, size = {ori_shape.nbDims - 1}, stride = {1}; - weights_shape_tensor = engine->network()->addSlice(*ori_shape_tensor, - sizes_to_nvdim(start), - sizes_to_nvdim(size), - sizes_to_nvdim(stride))->getOutput(0); - } - // if gamma exist - if (maybe_weight.isTensor()) { - torch::Tensor gamma = maybe_weight.toTensor(); - nvinfer1::ITensor* gamma_tensor = tensor_to_const(engine, gamma); - nvinfer1::ITensor* gamma_tensor_expand = expand_gamma_beta(engine, - gamma_tensor, - sizes_to_nvdim(weights_shuffle_dims_vec), - weights_shape_tensor, - sizes_to_nvdim(weights_dims_vec), - is_dynamic, - layer_info(node) + "_ISliceLayer_for_gamma"); - scale_l = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - cat_out, - gamma_tensor_expand, - (layer_info(node) + "_prod_for_scale").c_str()); - } - // if beta exist - if (maybe_bias.isTensor()) { - torch::Tensor ori_beta = maybe_bias.toTensor(); - nvinfer1::ITensor* beta_tensor = tensor_to_const(engine, ori_beta); - nvinfer1::ITensor* beta_tensor_expand = expand_gamma_beta(engine, - beta_tensor, - sizes_to_nvdim(weights_shuffle_dims_vec), - weights_shape_tensor, - sizes_to_nvdim(weights_dims_vec), - is_dynamic, - layer_info(node) + "_ISliceLayer_for_beta"); - if (scale_l == nullptr) { - shift_l = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - cat_out, - beta_tensor_expand, - (layer_info(node) + "_sum_for_shift").c_str()); - - } else { - shift_l = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - scale_l->getOutput(0), - beta_tensor_expand, - (layer_info(node) + "_sum_for_shift").c_str()); - } - nvinfer1::ITensor* shift_l_out = shift_l->getOutput(0); - engine->context().set_tensor(node->outputs()[0], shift_l_out); - LOG(INFO) << "Output tensor shape: " << shift_l_out->getDimensions(); - } else { - nvinfer1::ITensor* scale_l_out = scale_l->getOutput(0); - engine->context().set_tensor(node->outputs()[0], scale_l_out); - LOG(INFO) << "Output tensor shape: " << scale_l_out->getDimensions(); - - } - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, GroupNormConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/group_norm.h b/poros/poros/converter/gpu/group_norm.h deleted file mode 100644 index c7057cbc79a..00000000000 --- a/poros/poros/converter/gpu/group_norm.h +++ /dev/null @@ -1,56 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file group_norm.h -* @author tianshaoqing@baidu.com -* @date Fri Jan 21 15:28:37 CST 2022 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class GroupNormConverter : public GpuConverter { -public: - GroupNormConverter() {} - virtual ~GroupNormConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - bool converter_old(TensorrtEngine* engine, const torch::jit::Node *node); - - virtual const std::vector schema_string() { - return {"aten::group_norm(Tensor input, int num_groups, Tensor? weight=None, Tensor? bias=None, float eps=1e-05, bool cudnn_enabled=True) -> Tensor"}; - } - - virtual const std::vector node_kind() { - return {torch::jit::aten::group_norm}; - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/interpolate.cpp b/poros/poros/converter/gpu/interpolate.cpp deleted file mode 100644 index 2a98b090683..00000000000 --- a/poros/poros/converter/gpu/interpolate.cpp +++ /dev/null @@ -1,599 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -// Part of the following code in this file refs to -// https://github.com/pytorch/TensorRT/blob/master/core/conversion/converters/impl/interpolate.cpp -// -// Copyright (c) 2020-present, NVIDIA CORPORATION. All rights reserved. -// Copyright (c) Meta Platforms, Inc. and affiliates. -// Licensed under the 3-Clause BSD License - -/** -* @file interpolate.cpp -* @author tianjinjin@baidu.com -* @date Mon Mar 8 11:36:11 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/interpolate.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/* - * Helper functions - */ -void create_plugin(TensorrtEngine* engine, - const torch::jit::Node* node, - nvinfer1::ITensor* in, - const char* name, - std::vector in_shape, - std::vector out_shape, - std::vector out_size, - std::vector scales, - std::string mode, - bool align_corners, - bool use_scales = false) { - LOG(WARNING) << "Interpolation layer will be run through ATen, not TensorRT. " - << "Performance may be lower than expected"; - nvinfer1::PluginFieldCollection fc; - std::vector f; - std::vector in_shape_casted(in_shape.begin(), in_shape.end()); - f.emplace_back(nvinfer1::PluginField( - "in_shape", in_shape_casted.data(), nvinfer1::PluginFieldType::kINT32, in_shape.size())); - - std::vector out_shape_casted(out_shape.begin(), out_shape.end()); - f.emplace_back(nvinfer1::PluginField( - "out_shape", out_shape_casted.data(), nvinfer1::PluginFieldType::kINT32, out_shape.size())); - - std::vector out_size_casted(out_size.begin(), out_size.end()); - f.emplace_back(nvinfer1::PluginField( - "out_size", out_size_casted.data(), nvinfer1::PluginFieldType::kINT32, out_size.size())); - - f.emplace_back(nvinfer1::PluginField( - "scales", scales.data(), nvinfer1::PluginFieldType::kFLOAT64, scales.size())); - f.emplace_back(nvinfer1::PluginField( - "mode", &mode, nvinfer1::PluginFieldType::kCHAR, 1)); - - int32_t align_corners_casted = static_cast(align_corners); - f.emplace_back(nvinfer1::PluginField( - "align_corners", &align_corners_casted, nvinfer1::PluginFieldType::kINT32, 1)); - - int32_t use_scales_casted = static_cast(use_scales); - f.emplace_back(nvinfer1::PluginField( - "use_scales", &use_scales_casted, nvinfer1::PluginFieldType::kINT32, 1)); - - fc.nbFields = f.size(); - fc.fields = f.data(); - auto creator = getPluginRegistry()->getPluginCreator("Interpolate", "1", ""); - auto interpolate_plugin = creator->createPlugin(name, &fc); - - auto resize_layer = engine->network()->addPluginV2( - reinterpret_cast(&in), 1, *interpolate_plugin); - POROS_CHECK(resize_layer, "Unable to create interpolation plugin from node" << *node); - resize_layer->setName((layer_info(node) + "_plugin_Interpolate").c_str()); - - engine->context().set_tensor(node->outputs()[0], resize_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << resize_layer->getOutput(0)->getDimensions(); -} - -void resize_layer_size(TensorrtEngine* engine, - const torch::jit::Node* node, - nvinfer1::ITensor* in, - std::vector out_shape, - std::vector scales, - nvinfer1::ResizeMode mode, - bool align_corners = false) { - POROS_CHECK((out_shape.size() > 0) ^ (scales.size() > 0), "only one of out_shape or scales should be defined"); - auto resize_layer = engine->network()->addResize(*in); - POROS_CHECK(resize_layer, "Unable to create interpolation (resizing) layer from node" << *node); - - if (out_shape.size() > 0) { - auto th_dynamic_shape_mask = torch::zeros(out_shape.size(), torch::kInt32); - auto th_static_shape_mask = torch::zeros(out_shape.size(), torch::kInt32); - for (size_t idx = 0; idx < out_shape.size(); ++idx) { - if (out_shape[idx] == -1) { - th_dynamic_shape_mask[idx] = 1; - } else { - th_static_shape_mask[idx] = out_shape[idx]; - } - } - - auto dynamic_shape_mask = tensor_to_const(engine, th_dynamic_shape_mask); - auto static_shape_mask = tensor_to_const(engine, th_static_shape_mask); - auto input_shape = engine->network()->addShape(*in)->getOutput(0); - auto dynamic_shape = engine->network()->addElementWise( - *input_shape, *dynamic_shape_mask, nvinfer1::ElementWiseOperation::kPROD)->getOutput(0); - auto target_output_shape = engine->network()->addElementWise( - *dynamic_shape, *static_shape_mask, nvinfer1::ElementWiseOperation::kSUM)->getOutput(0); - resize_layer->setInput(1, *target_output_shape); - } else { - resize_layer->setScales(scales.data(), scales.size()); - if (align_corners) { - LOG(WARNING) << "interpolate with align_corners and scale_factor works differently in TensorRT and PyTorch."; - } - } - - resize_layer->setResizeMode(mode); - resize_layer->setName((layer_info(node) + "_IResizeLayer").c_str()); -#if NV_TENSORRT_MAJOR < 8 - resize_layer->setAlignCorners(align_corners); -#else - if (align_corners) { - resize_layer->setCoordinateTransformation(nvinfer1::ResizeCoordinateTransformation::kALIGN_CORNERS); - } -#endif - engine->context().set_tensor(node->outputs()[0], resize_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << resize_layer->getOutput(0)->getDimensions(); -} - -/* -"aten::upsample_nearest1d(Tensor self, int[1] output_size, float? scales=None) -> Tensor", -"aten::upsample_nearest1d.vec(Tensor input, int[]? output_size, float[]? scale_factors) -> Tensor", -*/ -bool UnsampleNearest1DConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for UnsampleNearest1DConverter is not Tensor as expected"); - - //extract in - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - auto in_shape = nvdim_to_sizes(in->getDimensions()); - - auto maybe_outsize = engine->context().get_constant(inputs[1]); - auto maybe_scales = engine->context().get_constant(inputs[2]); - if (maybe_outsize.isNone() && maybe_scales.isNone()) { - POROS_THROW_ERROR("Unable to convert node: " << node_info(node) - << "\nOne of output_size or scale_factors should be defined"); - } - - if (!maybe_scales.isNone()) { - float scale = 0.0f; - // Case 1: user uses scales - if (maybe_scales.isDouble()) { - scale = maybe_scales.toDouble(); - } else { // maybe_scales.isDoubleList() - auto scale_factors = maybe_scales.toDoubleList(); - POROS_ASSERT(scale_factors.size() == 1, "Number of scale factors should match the input size"); - scale = scale_factors[0]; - } - std::vector padded_scales(in_shape.size(), 1); - padded_scales[padded_scales.size() - 1] = scale; - resize_layer_size(engine, node, in, {}, padded_scales, nvinfer1::ResizeMode::kNEAREST); - } else { - // Case 2: user uses output size - auto output_size = maybe_outsize.toIntList(); - auto out_size = nvdim_to_sizes(sizes_to_nvdim(output_size)); - POROS_ASSERT(out_size.size() == 1, "aten::upsample_nearest1d input Tensor and output size dimension mismatch"); - auto out_shape = in_shape; - std::copy(out_size.begin(), out_size.end(), out_shape.begin() + (in_shape.size() - out_size.size())); - resize_layer_size(engine, node, in, out_shape, {}, nvinfer1::ResizeMode::kNEAREST); - } - return true; -} - -/* -"aten::upsample_nearest2d(Tensor self, int[2] output_size, float? scales_h=None, float? scales_w=None) -> Tensor", -"aten::upsample_nearest2d.vec(Tensor input, int[]? output_size, float[]? scale_factors) -> Tensor", -*/ -bool UnsampleNearest2DConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for UnsampleNearest2DConverter is not Tensor as expected"); - - //extract in - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - auto in_shape = nvdim_to_sizes(in->getDimensions()); - - auto maybe_outsize = engine->context().get_constant(inputs[1]); - float scale_h = 0.0f; - float scale_w = 0.0f; - - if (inputs.size() == 4) { - auto maybe_scales_h = engine->context().get_constant(inputs[2]); - auto maybe_scales_w = engine->context().get_constant(inputs[3]); - if (maybe_outsize.isNone() && (maybe_scales_h.isNone() || maybe_scales_w.isNone())) { - POROS_THROW_ERROR("Unable to convert node: " << node_info(node) - << "\nOne of output_size or scales should be defined"); - } - if (!maybe_scales_h.isNone() && !maybe_scales_w.isNone()) { - // Case 1: user uses scales - scale_h = maybe_scales_h.toDouble(); - scale_w = maybe_scales_w.toDouble(); - } - } else { //(inputs_size() == 3) - auto maybe_scale_factors = engine->context().get_constant(inputs[2]); - if (maybe_outsize.isNone() && maybe_scale_factors.isNone()) { - POROS_THROW_ERROR("Unable to convert node: " << node_info(node) - << "\nOne of output_size or scale_factors should be defined"); - } - - if (!maybe_scale_factors.isNone()) { - // Case 1: user uses scales - auto scale_factors = maybe_scale_factors.toDoubleList(); - POROS_ASSERT(scale_factors.size() == 2, "Number of scale factors should match the input size"); - scale_h = scale_factors[0]; - scale_w = scale_factors[1]; - } - } - - if (!engine->context().get_constant(inputs[2]).isNone()) { - std::vector padded_scales(in_shape.size(), 1); - padded_scales[padded_scales.size() - 2] = scale_h; - padded_scales[padded_scales.size() - 1] = scale_w; - resize_layer_size(engine, node, in, {}, padded_scales, nvinfer1::ResizeMode::kNEAREST); - } else { - // Case 2: user uses output size - auto output_size = maybe_outsize.toIntList(); - auto out_size = nvdim_to_sizes(sizes_to_nvdim(output_size)); - POROS_ASSERT(out_size.size() == 2, "aten::upsample_nearest2d input Tensor and output size dimension mismatch"); - auto out_shape = in_shape; - std::copy(out_size.begin(), out_size.end(), out_shape.begin() + (in_shape.size() - out_size.size())); - resize_layer_size(engine, node, in, out_shape, {}, nvinfer1::ResizeMode::kNEAREST); - } - return true; -} - -/* -"aten::upsample_nearest3d(Tensor self, int[3] output_size, float? scales_d=None, float? scales_h=None, float? scales_w=None) -> Tensor", -"aten::upsample_nearest3d.vec(Tensor input, int[]? output_size, float[]? scale_factors) -> Tensor", -*/ -bool UnsampleNearest3DConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for UnsampleNearest3DConverter is not Tensor as expected"); - - //extract in - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - auto in_shape = nvdim_to_sizes(in->getDimensions()); - - auto maybe_outsize = engine->context().get_constant(inputs[1]); - float scale_d = 0.0f; - float scale_h = 0.0f; - float scale_w = 0.0f; - - if (inputs.size() == 5) { - auto maybe_scales_d = engine->context().get_constant(inputs[2]); - auto maybe_scales_h = engine->context().get_constant(inputs[3]); - auto maybe_scales_w = engine->context().get_constant(inputs[4]); - if (maybe_outsize.isNone() && (maybe_scales_d.isNone() || - maybe_scales_h.isNone() || maybe_scales_w.isNone())) { - POROS_THROW_ERROR("Unable to convert node: " << node_info(node) - << "\nOne of output_size or scales should be defined"); - } - if (!maybe_scales_d.isNone() && !maybe_scales_h.isNone() && !maybe_scales_w.isNone()) { - // Case 1: user uses scales - scale_d = maybe_scales_d.toDouble(); - scale_h = maybe_scales_h.toDouble(); - scale_w = maybe_scales_w.toDouble(); - } - } else { //(inputs_size() == 3) - auto maybe_scale_factors = engine->context().get_constant(inputs[2]); - if (maybe_outsize.isNone() && maybe_scale_factors.isNone()) { - POROS_THROW_ERROR("Unable to convert node: " << node_info(node) - << "\nOne of output_size or scale_factors should be defined"); - } - - if (!maybe_scale_factors.isNone()) { - // Case 1: user uses scales - auto scale_factors = maybe_scale_factors.toDoubleList(); - POROS_ASSERT(scale_factors.size() == 3, "Number of scale factors should match the input size"); - scale_d = scale_factors[0]; - scale_h = scale_factors[1]; - scale_w = scale_factors[2]; - } - } - - if (!engine->context().get_constant(inputs[2]).isNone()) { - std::vector padded_scales(in_shape.size(), 1); - padded_scales[padded_scales.size() - 3] = scale_d; - padded_scales[padded_scales.size() - 2] = scale_h; - padded_scales[padded_scales.size() - 1] = scale_w; - resize_layer_size(engine, node, in, {}, padded_scales, nvinfer1::ResizeMode::kNEAREST); - } else { - // Case 2: user uses output size - auto output_size = maybe_outsize.toIntList(); - auto out_size = nvdim_to_sizes(sizes_to_nvdim(output_size)); - POROS_ASSERT(out_size.size() == 3, "aten::upsample_nearest3d input Tensor and output size dimension mismatch"); - auto out_shape = in_shape; - std::copy(out_size.begin(), out_size.end(), out_shape.begin() + (in_shape.size() - out_size.size())); - resize_layer_size(engine, node, in, out_shape, {}, nvinfer1::ResizeMode::kNEAREST); - } - return true; -} - -/* -"aten::upsample_linear1d(Tensor self, int[1] output_size, bool align_corners, float? scales=None) -> Tensor", -"aten::upsample_linear1d.vec(Tensor input, int[]? output_size, bool align_corners, float[]? scale_factors) -> Tensor", -*/ -bool UnsampleLinear1DConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for UnsampleLinear1DConverter is not Tensor as expected"); - - //extract in - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - auto in_shape = nvdim_to_sizes(in->getDimensions()); - //extract align_corners - auto align_corners = engine->context().get_constant(inputs[2]).toBool(); - - auto maybe_outsize = engine->context().get_constant(inputs[1]); - auto maybe_scales = engine->context().get_constant(inputs[3]); - if (maybe_outsize.isNone() && maybe_scales.isNone()) { - POROS_THROW_ERROR("Unable to convert node: " << node_info(node) - << "\nOne of output_size or scales should be defined"); - } - - if (!maybe_scales.isNone()) { - // Case 1: user uses scales - float scale = 0.0f; - if (maybe_scales.isDouble()) { - scale = maybe_scales.toDouble(); - } else { //maybe_scales.isDoubleList() - auto scale_factors = maybe_scales.toDoubleList(); - POROS_ASSERT(scale_factors.size() == 1, "Number of scale factors should match the input size"); - scale = scale_factors[0]; - } - std::vector padded_scales(in_shape.size(), 1); - padded_scales[padded_scales.size() - 1] = scale; -#if NV_TENSORRT_MAJOR < 7 || (NV_TENSORRT_MAJOR == 7 && NV_TENSORRT_MINOR < 1) // IF TRT VERSION <= 7.0 - if (!align_corners) { - POROS_THROW_ERROR("Unable to convert node: " << node_info(node) - << "\nupsample_linear1d only supports align_corner with TensorRT <= 7.0."); - } else { - resize_layer_size(engine, node, in, {}, padded_scales, nvinfer1::ResizeMode::kLINEAR, true); - } -#else - auto is_dynamic_shape = PorosGlobalContext::instance().get_poros_options().is_dynamic; - POROS_CHECK(!(align_corners && is_dynamic_shape), "Poros currently does not support the compilation of dynamc engines" - << "from code using using PyTorch [bi/tri]linear interpolation via scale factor and align_corners=True"); - if (align_corners) { - // Align corners and scale factor behave slightly different together in TRT and PyTorch so run the - // layer in ATen to maintain consistancy between TRTorch and PyTorch - // https://pytorch.org/docs/stable/nn.functional.html#torch.nn.functional.interpolate - create_plugin(engine, node, in, "linear1d", in_shape, {}, {}, {scale}, std::string("linear"), align_corners, true); - } else { - resize_layer_size(engine, node, in, {}, padded_scales, nvinfer1::ResizeMode::kLINEAR, align_corners); - } -#endif - } else { - // Case 2: user uses output size - auto output_size = maybe_outsize.toIntList(); - auto out_size = nvdim_to_sizes(sizes_to_nvdim(output_size)); - POROS_ASSERT(out_size.size() == 1, "aten::upsample_linear1d input Tensor and output size dimension mismatch"); - auto out_shape = in_shape; - std::copy(out_size.begin(), out_size.end(), out_shape.begin() + (in_shape.size() - out_size.size())); -#if NV_TENSORRT_MAJOR < 7 || (NV_TENSORRT_MAJOR == 7 && NV_TENSORRT_MINOR < 1) // IF TRT VERSION <= 7.0 - if (!align_corners) { - create_plugin(engine, node, in, "linear1d", in_shape, out_shape, out_size, {}, std::string("linear"), align_corners); - } else { - resize_layer_size(engine, node, in, out_shape, {}, nvinfer1::ResizeMode::kLINEAR, true); - } -#else - resize_layer_size(engine, node, in, out_shape, {}, nvinfer1::ResizeMode::kLINEAR, align_corners); -#endif - } - return true; -} - -/* -"aten::upsample_bilinear2d(Tensor self, int[2] output_size, bool align_corners, float? scales_h=None, float? scales_w=None) -> Tensor", -"aten::upsample_bilinear2d.vec(Tensor input, int[]? output_size, bool align_corners, float[]? scale_factors) -> Tensor", -*/ -bool UnsampleBilinear2DConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for UnsampleBilinear2DConverter is not Tensor as expected"); - - //extract in - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - auto in_shape = nvdim_to_sizes(in->getDimensions()); - //extract align_corners - auto align_corners = engine->context().get_constant(inputs[2]).toBool(); - - auto maybe_outsize = engine->context().get_constant(inputs[1]); - float scale_h = 0.0f; - float scale_w = 0.0f; - - if (inputs.size() == 5) { - auto maybe_scales_h = engine->context().get_constant(inputs[3]); - auto maybe_scales_w = engine->context().get_constant(inputs[4]); - if (maybe_outsize.isNone() && (maybe_scales_h.isNone() || maybe_scales_w.isNone())) { - POROS_THROW_ERROR("Unable to convert node: " << node_info(node) - << "\nOne of output_size or scales should be defined"); - } - if (!maybe_scales_h.isNone() && !maybe_scales_w.isNone()) { - // Case 1: user uses scales - scale_h = maybe_scales_h.toDouble(); - scale_w = maybe_scales_w.toDouble(); - } - } else { //(inputs_size() == 4) - auto maybe_scale_factors = engine->context().get_constant(inputs[3]); - if (maybe_outsize.isNone() && maybe_scale_factors.isNone()) { - POROS_THROW_ERROR("Unable to convert node: " << node_info(node) - << "\nOne of output_size or scale_factors should be defined"); - } - if (!maybe_scale_factors.isNone()) { - // Case 1: user uses scales - auto scale_factors = maybe_scale_factors.toDoubleList(); - POROS_ASSERT(scale_factors.size() == 2, "Number of scale factors should match the input size"); - scale_h = scale_factors[0]; - scale_w = scale_factors[1]; - } - } - - if (!engine->context().get_constant(inputs[3]).isNone()) { - std::vector padded_scales(in_shape.size(), 1); - padded_scales[padded_scales.size() - 2] = scale_h; - padded_scales[padded_scales.size() - 1] = scale_w; -#if NV_TENSORRT_MAJOR < 7 || (NV_TENSORRT_MAJOR == 7 && NV_TENSORRT_MINOR < 1) // IF TRT VERSION <= 7.0 - if (!align_corners) { - POROS_THROW_ERROR("Unable to convert node: " << node_info(node) - << "\nupsample_linear1d only supports align_corner with TensorRT <= 7.0."); - } else { - resize_layer_size(engine, node, in, {}, padded_scales, nvinfer1::ResizeMode::kLINEAR, true); - } -#else - auto is_dynamic_shape = PorosGlobalContext::instance().get_poros_options().is_dynamic; - POROS_CHECK(!(align_corners && is_dynamic_shape), "Poros currently does not support the compilation of dynamc engines" - << "from code using using PyTorch [bi/tri]linear interpolation via scale factor and align_corners=True"); - if (align_corners) { - // Align corners and scale factor behave slightly different together in TRT and PyTorch so run the - // layer in ATen to maintain consistancy between TRTorch and PyTorch - // https://pytorch.org/docs/stable/nn.functional.html#torch.nn.functional.interpolate - create_plugin(engine, node, in, "bilinear2d", in_shape, {}, {}, {scale_h, scale_w}, std::string("bilinear"), align_corners, true); - } else { - resize_layer_size(engine, node, in, {}, padded_scales, nvinfer1::ResizeMode::kLINEAR, align_corners); - } -#endif - } else { - // Case 2: user uses output size - auto output_size = maybe_outsize.toIntList(); - auto out_size = nvdim_to_sizes(sizes_to_nvdim(output_size)); - POROS_ASSERT(out_size.size() == 2, "aten::upsample_bilinear2d input Tensor and output size dimension mismatch"); - auto out_shape = in_shape; - std::copy(out_size.begin(), out_size.end(), out_shape.begin() + (in_shape.size() - out_size.size())); -#if NV_TENSORRT_MAJOR < 7 || (NV_TENSORRT_MAJOR == 7 && NV_TENSORRT_MINOR < 1) // IF TRT VERSION <= 7.0 - if (!align_corners) { - create_plugin(engine, node, in, "bilinear2d", in_shape, out_shape, out_size, {}, std::string("bilinear"), align_corners); - } else { - resize_layer_size(engine, node, in, out_shape, {}, nvinfer1::ResizeMode::kLINEAR, true); - } -#else - resize_layer_size(engine, node, in, out_shape, {}, nvinfer1::ResizeMode::kLINEAR, align_corners); -#endif - } - return true; -} - -/* -"aten::upsample_trilinear3d(Tensor self, int[3] output_size, bool align_corners, float? scales_d=None, float? scales_h=None, float? scales_w=None) -> Tensor", -"aten::upsample_trilinear3d.vec(Tensor input, int[]? output_size, bool align_corners, float[]? scale_factors) -> Tensor", -*/ -bool UnsampleTrilinear3DConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for UnsampleTrilinear3DConverter is not Tensor as expected"); - - //extract in - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - auto in_shape = nvdim_to_sizes(in->getDimensions()); - //extract align_corners - auto align_corners = engine->context().get_constant(inputs[2]).toBool(); - - auto maybe_outsize = engine->context().get_constant(inputs[1]); - float scale_d = 0.0f; - float scale_h = 0.0f; - float scale_w = 0.0f; - - if (inputs.size() == 6) { - auto maybe_scales_d = engine->context().get_constant(inputs[3]); - auto maybe_scales_h = engine->context().get_constant(inputs[4]); - auto maybe_scales_w = engine->context().get_constant(inputs[5]); - if (maybe_outsize.isNone() && (maybe_scales_h.isNone() - || maybe_scales_w.isNone() || maybe_scales_d.isNone())) { - POROS_THROW_ERROR("Unable to convert node: " << node_info(node) - << "\nOne of output_size or scales should be defined"); - } - if (!maybe_scales_h.isNone() && !maybe_scales_w.isNone() && maybe_scales_d.isNone()) { - // Case 1: user uses scales - scale_d = maybe_scales_d.toDouble(); - scale_h = maybe_scales_h.toDouble(); - scale_w = maybe_scales_w.toDouble(); - } - } else { //(inputs_size() == 4) - auto maybe_scale_factors = engine->context().get_constant(inputs[3]); - if (maybe_outsize.isNone() && maybe_scale_factors.isNone()) { - POROS_THROW_ERROR("Unable to convert node: " << node_info(node) - << "\nOne of output_size or scale_factors should be defined"); - } - if (!maybe_scale_factors.isNone()) { - // Case 1: user uses scales - auto scale_factors = maybe_scale_factors.toDoubleList(); - POROS_ASSERT(scale_factors.size() == 3, "Number of scale factors should match the input size"); - scale_d = scale_factors[0]; - scale_h = scale_factors[1]; - scale_w = scale_factors[2]; - } - } - - if (!engine->context().get_constant(inputs[3]).isNone()) { - std::vector padded_scales(in_shape.size(), 1); - padded_scales[padded_scales.size() - 3] = scale_d; - padded_scales[padded_scales.size() - 2] = scale_h; - padded_scales[padded_scales.size() - 1] = scale_w; -#if NV_TENSORRT_MAJOR < 7 || (NV_TENSORRT_MAJOR == 7 && NV_TENSORRT_MINOR < 1) // IF TRT VERSION <= 7.0 - if (!align_corners) { - POROS_THROW_ERROR("Unable to convert node: " << node_info(node) - << "\nupsample_linear1d only supports align_corner with TensorRT <= 7.0."); - } else { - resize_layer_size(engine, node, in, {}, padded_scales, nvinfer1::ResizeMode::kLINEAR, true); - } -#else - auto is_dynamic_shape = PorosGlobalContext::instance().get_poros_options().is_dynamic; - POROS_CHECK(!(align_corners && is_dynamic_shape), "Poros currently does not support the compilation of dynamc engines" - << "from code using using PyTorch [bi/tri]linear interpolation via scale factor and align_corners=True"); - if (align_corners) { - // Align corners and scale factor behave slightly different together in TRT and PyTorch so run the - // layer in ATen to maintain consistancy between TRTorch and PyTorch - // https://pytorch.org/docs/stable/nn.functional.html#torch.nn.functional.interpolate - create_plugin(engine, node, in, "trilinear3d", in_shape, {}, {}, {scale_d, scale_h, scale_w}, std::string("trilinear"), align_corners, true); - } else { - resize_layer_size(engine, node, in, {}, padded_scales, nvinfer1::ResizeMode::kLINEAR, align_corners); - } -#endif - } else { - // Case 2: user uses output size - auto output_size = maybe_outsize.toIntList(); - auto out_size = nvdim_to_sizes(sizes_to_nvdim(output_size)); - POROS_ASSERT(out_size.size() == 3, "aten::upsample_trilinear3d input Tensor and output size dimension mismatch"); - auto out_shape = in_shape; - std::copy(out_size.begin(), out_size.end(), out_shape.begin() + (in_shape.size() - out_size.size())); -#if NV_TENSORRT_MAJOR < 7 || (NV_TENSORRT_MAJOR == 7 && NV_TENSORRT_MINOR < 1) // IF TRT VERSION <= 7.0 - if (!align_corners) { - create_plugin(engine, node, in, "trilinear3d", in_shape, out_shape, out_size, {}, std::string("trilinear"), align_corners); - } else { - resize_layer_size(engine, node, in, out_shape, {}, nvinfer1::ResizeMode::kLINEAR, true); - } -#else - resize_layer_size(engine, node, in, out_shape, {}, nvinfer1::ResizeMode::kLINEAR, align_corners); -#endif - } - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, UnsampleNearest1DConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, UnsampleNearest2DConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, UnsampleNearest3DConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, UnsampleLinear1DConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, UnsampleBilinear2DConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, UnsampleTrilinear3DConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/interpolate.h b/poros/poros/converter/gpu/interpolate.h deleted file mode 100644 index 90ceca75fdd..00000000000 --- a/poros/poros/converter/gpu/interpolate.h +++ /dev/null @@ -1,164 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file interpolate.h -* @author tianjinjin@baidu.com -* @date Mon Aug 16 12:26:28 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class UnsampleNearest1DConverter : public GpuConverter { -public: - UnsampleNearest1DConverter() {} - virtual ~UnsampleNearest1DConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::upsample_nearest1d(Tensor self, int[1] output_size, float? scales=None) -> Tensor", - "aten::upsample_nearest1d.vec(Tensor input, int[]? output_size, float[]? scale_factors) -> Tensor", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::upsample_nearest1d.out(Tensor self, int[1] output_size, float? scales=None, *, Tensor(a!) out) -> Tensor(a!)", - * **/ - const std::vector node_kind() { - return {torch::jit::aten::upsample_nearest1d}; - } -}; - -class UnsampleNearest2DConverter : public GpuConverter { -public: - UnsampleNearest2DConverter() {} - virtual ~UnsampleNearest2DConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::upsample_nearest2d(Tensor self, int[2] output_size, float? scales_h=None, float? scales_w=None) -> Tensor", - "aten::upsample_nearest2d.vec(Tensor input, int[]? output_size, float[]? scale_factors) -> Tensor", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::upsample_nearest2d.out(Tensor self, int[2] output_size, float? scales_h=None, float? scales_w=None, *, Tensor(a!) out) -> Tensor(a!)", - * **/ - const std::vector node_kind() { - return {torch::jit::aten::upsample_nearest2d}; - } -}; - -class UnsampleNearest3DConverter : public GpuConverter { -public: - UnsampleNearest3DConverter() {} - virtual ~UnsampleNearest3DConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::upsample_nearest3d(Tensor self, int[3] output_size, float? scales_d=None, float? scales_h=None, float? scales_w=None) -> Tensor", - "aten::upsample_nearest3d.vec(Tensor input, int[]? output_size, float[]? scale_factors) -> Tensor", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::upsample_nearest3d.out(Tensor self, int[3] output_size, float? scales_d=None, float? scales_h=None, float? scales_w=None, *, Tensor(a!) out) -> Tensor(a!)", - * **/ - const std::vector node_kind() { - return {torch::jit::aten::upsample_nearest3d}; - } -}; - -class UnsampleLinear1DConverter : public GpuConverter { -public: - UnsampleLinear1DConverter() {} - virtual ~UnsampleLinear1DConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::upsample_linear1d(Tensor self, int[1] output_size, bool align_corners, float? scales=None) -> Tensor", - "aten::upsample_linear1d.vec(Tensor input, int[]? output_size, bool align_corners, float[]? scale_factors) -> Tensor", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::upsample_linear1d.out(Tensor self, int[1] output_size, bool align_corners, float? scales=None, *, Tensor(a!) out) -> Tensor(a!)", - * **/ - const std::vector node_kind() { - return {torch::jit::aten::upsample_linear1d}; - } -}; - -class UnsampleBilinear2DConverter : public GpuConverter { -public: - UnsampleBilinear2DConverter() {} - virtual ~UnsampleBilinear2DConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::upsample_bilinear2d(Tensor self, int[2] output_size, bool align_corners, float? scales_h=None, float? scales_w=None) -> Tensor", - "aten::upsample_bilinear2d.vec(Tensor input, int[]? output_size, bool align_corners, float[]? scale_factors) -> Tensor", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::upsample_bilinear2d.out(Tensor self, int[2] output_size, bool align_corners, float? scales_h=None, float? scales_w=None, *, Tensor(a!) out) -> Tensor(a!)", - * **/ - const std::vector node_kind() { - return {torch::jit::aten::upsample_bilinear2d}; - } -}; - -class UnsampleTrilinear3DConverter : public GpuConverter { -public: - UnsampleTrilinear3DConverter() {} - virtual ~UnsampleTrilinear3DConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::upsample_trilinear3d(Tensor self, int[3] output_size, bool align_corners, float? scales_d=None, float? scales_h=None, float? scales_w=None) -> Tensor", - "aten::upsample_trilinear3d.vec(Tensor input, int[]? output_size, bool align_corners, float[]? scale_factors) -> Tensor", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::upsample_trilinear3d.out(Tensor self, int[3] output_size, bool align_corners, float? scales_d=None, float? scales_h=None, float? scales_w=None, *, Tensor(a!) out) -> Tensor(a!)", - * **/ - const std::vector node_kind() { - return {torch::jit::aten::upsample_trilinear3d}; - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/layer_norm.cpp b/poros/poros/converter/gpu/layer_norm.cpp deleted file mode 100644 index 16cd2296c15..00000000000 --- a/poros/poros/converter/gpu/layer_norm.cpp +++ /dev/null @@ -1,198 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file layer_norm.cpp -* @author tianjinjin@baidu.com -* @date Fri Aug 20 15:28:37 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/layer_norm.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/converter/gpu/weight.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/* -aten::layer_norm(Tensor input, -int[] normalized_shape, -Tensor? weight=None, -Tensor? bias=None, -float eps=1e-05, -bool cudnn_enable=True) -> Tensor -*/ -bool LayerNormConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 6), "invaid inputs size for LayerNormConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for LayerNormConverter is not Tensor as expected"); - // weight & bias - POROS_CHECK_TRUE((inputs[2]->node()->kind() == torch::jit::prim::Constant), - "input[2] for LayerNormConverter is not come from prim::Constant as expected"); - POROS_CHECK_TRUE((inputs[3]->node()->kind() == torch::jit::prim::Constant), - "input[3] for LayerNormConverter is not come from prim::Constant as expected"); - - nvinfer1::ITensor* input = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((input != nullptr), "Unable to init input tensor for node: " << *node); - nvinfer1::Dims orig_shape = input->getDimensions(); - std::vector shape = nvdim_to_sizes(orig_shape); - - /* Layer_Norm normalizes over last N dimensions. - normalizaed_shape could be (C,H,W), (H,W), or (W). */ - c10::List normalized_shape = (engine->context().get_constant(inputs[1])).toIntList(); - std::vector normalized_shape_vec = nvdim_to_sizes(sizes_to_nvdim(normalized_shape)); - - // Unwrap eps. - double eps = (engine->context().get_constant(inputs[4])).toDouble(); - - // Set up axis_ask for E[x]. - uint32_t axis_mask = 0; - for (size_t i = 0; i < normalized_shape_vec.size(); i++) { - axis_mask |= 1 << (shape.size() - i - 1); - } - LOG(INFO) << "Axis Mask for E[x]" << std::bitset<32>(axis_mask); - - // E[x] - nvinfer1::IReduceLayer* mean_expected = engine->network()->addReduce(*input, - nvinfer1::ReduceOperation::kAVG, axis_mask, true); - POROS_CHECK(mean_expected, "Unable to create mean_expected from node: " << *node); - mean_expected->setName((layer_info(node) + "_IReduceLayer(mean_expected)").c_str()); - nvinfer1::ITensor* mean_expected_out = mean_expected->getOutput(0); - - // X-E[x] - nvinfer1::ILayer* sub = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUB, - input, - mean_expected_out, - (layer_info(node) + "_sub").c_str()); - POROS_CHECK(sub, "Unable to create Sub layer from node: " << *node); - nvinfer1::ITensor* xsubmean_out = sub->getOutput(0); - - // Variance = mean(pow(xsubmean,2)) - float pow_scalar = 2; - nvinfer1::ITensor* exponent = tensor_to_const(engine, torch::tensor({pow_scalar})); - nvinfer1::ILayer* pow = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPOW, - xsubmean_out, - exponent, - (layer_info(node) + "_pow").c_str()); - POROS_CHECK(pow, "Unable to create Pow layer from node: " << *node); - nvinfer1::ITensor* pow_out = pow->getOutput(0); - - nvinfer1::IReduceLayer* mean_var = engine->network()->addReduce(*pow_out, - nvinfer1::ReduceOperation::kAVG, axis_mask, true); - POROS_CHECK(mean_var, "Unable to create mean_var from node: " << *node); - mean_var->setName((layer_info(node) + "_IReduceLayer(mean_var)").c_str()); - nvinfer1::ITensor* mean_var_out = mean_var->getOutput(0); - - // Variance + eps - nvinfer1::ITensor* eps_tensor = tensor_to_const(engine, torch::tensor({eps})); - nvinfer1::ILayer* add = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - mean_var_out, - eps_tensor, - (layer_info(node) + "_sum").c_str()); - POROS_CHECK(add, "Unable to create Add layer from node: " << *node); - nvinfer1::ITensor* add_out = add->getOutput(0); - - // SQRT((Var + eps)) - nvinfer1::IUnaryLayer* sqrt = engine->network()->addUnary(*add_out, nvinfer1::UnaryOperation::kSQRT); - POROS_CHECK(sqrt, "Unable to create unary(sqrt) from node: " << *node); - sqrt->setName((layer_info(node) + "_IUnaryLayer").c_str()); - nvinfer1::ITensor* sqrt_out = sqrt->getOutput(0); - - // (x - E[x]) / sqrt((var + eps)) - nvinfer1::ILayer* div = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kDIV, - xsubmean_out, - sqrt_out, - (layer_info(node) + "_div").c_str()); - POROS_CHECK(div, "Unable to create div layer from node: " << *node); - nvinfer1::ITensor* div_out = div->getOutput(0); - - torch::jit::IValue maybe_weight = engine->context().get_constant(inputs[2]); - torch::jit::IValue maybe_bias = engine->context().get_constant(inputs[3]); - //when weight and bias setting is both None - if (!maybe_weight.isTensor() && !maybe_bias.isTensor()) { - engine->context().set_tensor(node->outputs()[0], div_out); - LOG(INFO) << "Output tensor shape: " << div_out->getDimensions(); - return true; - } - - /*------------------------------------------------------------ - * situation when weight or bias setting is not None - * ------------------------------------------------------------*/ - // Remove batch dimension from input shape for expand_size, which will - // be used to create weights for addScaleNd later. - - /** TODO: IS the first input size always are always be batch????? - * if not, this converter is not ok。 - * */ - - // Set up gamma and beta by tensor_to_const directly, - // boardcast will be done automatically when add_elementwise, so need not expand - nvinfer1::ILayer* scale_l = nullptr; - nvinfer1::ILayer* shift_l = nullptr; - if (maybe_weight.isTensor()) { - torch::Tensor gamma = maybe_weight.toTensor(); - nvinfer1::ITensor* gamma_tensor = tensor_to_const(engine, gamma); - scale_l = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - div_out, - gamma_tensor, - (layer_info(node) + "_prod_for_gamma").c_str()); - } - - if (maybe_bias.isTensor()) { - torch::Tensor ori_beta = maybe_bias.toTensor(); - nvinfer1::ITensor* beta_tensor = tensor_to_const(engine, ori_beta); - if (scale_l == nullptr) { - shift_l = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - div_out, - beta_tensor, - (layer_info(node) + "_sum_for_beta").c_str()); - - } else { - shift_l = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - scale_l->getOutput(0), - beta_tensor, - (layer_info(node) + "_sum_for_beta").c_str()); - } - nvinfer1::ITensor* shift_l_out = shift_l->getOutput(0); - engine->context().set_tensor(node->outputs()[0], shift_l_out); - LOG(INFO) << "Output tensor shape: " << shift_l_out->getDimensions(); - } else { - nvinfer1::ITensor* scale_l_out = scale_l->getOutput(0); - engine->context().set_tensor(node->outputs()[0], scale_l_out); - LOG(INFO) << "Output tensor shape: " << scale_l_out->getDimensions(); - - } - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, LayerNormConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/layer_norm.h b/poros/poros/converter/gpu/layer_norm.h deleted file mode 100644 index cd4d9504a24..00000000000 --- a/poros/poros/converter/gpu/layer_norm.h +++ /dev/null @@ -1,61 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file layer_norm.h -* @author tianjinjin@baidu.com -* @date Fri Aug 20 15:28:37 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class LayerNormConverter : public GpuConverter { -public: - LayerNormConverter() {} - virtual ~LayerNormConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - virtual const std::vector schema_string() { - return {"aten::layer_norm(Tensor input, int[] normalized_shape, Tensor? weight=None, Tensor? bias=None, float eps=1e-05, bool cudnn_enable=True) -> Tensor"}; - } - - virtual const std::vector node_kind() { - return {torch::jit::aten::layer_norm}; - } - - bool assign_schema_attr() { - bool result = true; - result &= assign_schema_attr_helper({{"aten::layer_norm(Tensor input, int[] normalized_shape, Tensor? weight=None, Tensor? bias=None, float eps=1e-05, bool cudnn_enable=True) -> Tensor", {1, 1}}}); - return result; - } - -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/linear.cpp b/poros/poros/converter/gpu/linear.cpp deleted file mode 100644 index cf9237ab84d..00000000000 --- a/poros/poros/converter/gpu/linear.cpp +++ /dev/null @@ -1,233 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file linear.cpp -* @author tianjinjin@baidu.com -* @date Fri Aug 20 17:21:44 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/linear.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/converter/gpu/weight.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/** aten::linear(Tensor input, Tensor weight, Tensor? bias=None) -> Tensor - * the implementation of aten::linear in pytorch is in file: aten/src/Aten/native/Linear.cpp - * the core function is like this: - * auto bias = bias_opt.has_value() - ? c10::MaybeOwned::borrowed(*bias_opt) - : c10::MaybeOwned::owned(c10::in_place); - if (input.dim() == 2 && bias->defined()) { - return at::addmm(*bias, input, weight.t()); - } - auto output = at::matmul(input, weight.t()); - if (bias->defined()) { - output.add_(*bias); - } - return output; -* we can refer to the implement of original pytorch. -* ****************************** -* %res = aten::linear(%input, %weight_0, %bias) -* try to converter matmul like below: -* -* %weight = aten::t(%weight_0) -* %mm = aten::matmul(%input, %weight) -* if %bias is None: -* return %mm -* else: -* if (input.dim == 2): -* %res = aten::add(%bias, %mm, 1) -* else: -* %res = aten::add(%mm, %bias, 1) -**/ -bool LinearConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 3), "invaid inputs size for LinearConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for LinearConverter is not Tensor as expected"); - - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - //auto origin_self_dim = self->getDimensions().nbDims; - - // handle weight - nvinfer1::ITensor* weight = nullptr; - bool need_trans = false; - auto maybe_weight = engine->context().get_constant(inputs[1]); - if (maybe_weight.isTensor()) { - //常量tensor - at::Tensor weight_t = maybe_weight.toTensor().t(); - int weight_rank = weight_t.sizes().size(); - //需要padding tensor 的情况,直接转置并padding完成后,再转constant_tensor, 避免命中tensorrt中constshuffle的tatic. - if (weight_rank < self->getDimensions().nbDims) { - at::Tensor padding_weight = weight_t; - for (int dim = weight_rank; dim < self->getDimensions().nbDims; ++dim) { - padding_weight = weight_t.unsqueeze(0); - } - weight = tensor_to_const(engine, padding_weight); - } else { - weight = tensor_to_const(engine, weight_t); - } - } else { - //weight 来自其他的tensor - weight = engine->context().get_tensor(inputs[1]); - if (weight->getDimensions().nbDims >= 2) { - need_trans = true; - } - /* //转置交给matmul, 不再自己shuffle实现。 - auto weight_before_trans = engine->context().get_tensor(inputs[1]); - auto weight_dims = weight_before_trans->getDimensions(); - if (weight_dims.nbDims < 2) { - weight = weight_before_trans; - } else { - //like aten::transpose(input, 0, 1) - auto shuffle_layer = engine->network()->addShuffle(*weight_before_trans); - POROS_CHECK(shuffle_layer, "Unable to create shuffle layer from node: " << *node); - nvinfer1::Permutation first_perm; - first_perm.order[0] = 1; - first_perm.order[1] = 0; - shuffle_layer->setFirstTranspose(first_perm); - shuffle_layer->setZeroIsPlaceholder(false); - shuffle_layer->setName((layer_info(node) + "_IShuffleLayer(weight_transpose)").c_str()); - weight = shuffle_layer->getOutput(0); - } */ - } - - // Ensure self and weight tensors have same nbDims by expanding the dimensions (from 0 axis) if - // necessary. - if (self->getDimensions().nbDims < weight->getDimensions().nbDims) { - self = add_padding(engine, node, self, weight->getDimensions().nbDims, false, false); - } else { - weight = add_padding(engine, node, weight, self->getDimensions().nbDims, false, false); - } - - nvinfer1::IMatrixMultiplyLayer* mm_layer = nullptr; - if (need_trans == true) { - mm_layer = engine->network()->addMatrixMultiply( - *self, nvinfer1::MatrixOperation::kNONE, *weight, nvinfer1::MatrixOperation::kTRANSPOSE); - } else { - mm_layer = engine->network()->addMatrixMultiply( - *self, nvinfer1::MatrixOperation::kNONE, *weight, nvinfer1::MatrixOperation::kNONE); - } - POROS_CHECK(mm_layer, "Unable to create matrix multiplication node: " << *node); - - auto bias = engine->context().get_tensor(inputs[2]); - /*-------------------------------------------------------------- - * bias is None situation - * -------------------------------------------------------------*/ - //bias is None situation return directly - if (bias == nullptr) { - mm_layer->setName((layer_info(node) + "_IMatrixMultiplyLayer").c_str()); - engine->context().set_tensor(node->outputs()[0], mm_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << mm_layer->getOutput(0)->getDimensions(); - return true; - } - - /*-------------------------------------------------------------- - * bias is not None situation - * -------------------------------------------------------------*/ - mm_layer->setName((layer_info(node) + "_IMatrixMultiplyLayer").c_str()); - - nvinfer1::ILayer* new_layer = nullptr; - // if (origin_self_dim == 2) { - // //TODO: ADD SOME FUNCTION HERE - // } else { - new_layer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - mm_layer->getOutput(0), - bias, - layer_info(node) + "_sum"); - //} - POROS_CHECK(new_layer, "Unable to create add layer from node: " << *node); - engine->context().set_tensor(node->outputs()[0], new_layer->getOutput(0)); - - LOG(INFO) << "Output tensor shape: " << new_layer->getOutput(0)->getDimensions(); - return true; -} - -//DEPRECATED: result do not match the pytorch output -bool LinearConverter::converter_fully_connect_version(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 3), "invaid inputs size for LinearConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for LinearConverter is not Tensor as expected"); - // weight & bias - POROS_CHECK_TRUE((inputs[1]->type()->isSubtypeOf(c10::TensorType::get())), - "input[1] for LinearConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[2]->node()->kind() == torch::jit::prim::Constant), - "input[2] for LinearConverter is not come from prim::Constant as expected"); - - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - auto shape = nvdim_to_sizes(in->getDimensions()); - LOG(INFO) << "Input tensor shape: " << in->getDimensions(); - - // PyTorch follows in: Nx*xIN, W: OUTxIN, B: OUT, out: Nx*xOUT - // TensorRT inserts a flatten in when following conv - POROS_ASSERT(shape.size() >= 2, - "aten::linear expects input tensors to be of shape [N,..., in features], but found input Tensor less than 2D"); - - if (shape.size() < 4) { - // Flatten - std::vector new_shape; - new_shape.push_back(shape[0]); - new_shape.push_back(1); - new_shape.push_back(1); - new_shape.push_back(nvdim_to_volume(sizes_to_nvdim(shape)) / shape[0]); - auto new_dims = sizes_to_nvdim(new_shape); - - LOG(INFO) << "Input shape is less than 4D got: " << sizes_to_nvdim(shape) - << ", inserting shuffle layer to reshape to 4D tensor shape: " << new_dims; - - auto in_shuffle = engine->network()->addShuffle(*in); - in_shuffle->setReshapeDimensions(new_dims); - in_shuffle->setName((layer_info(node) + "_IShuffleLayer").c_str()); - in = in_shuffle->getOutput(0); - } - - auto w_tensor = (engine->context().get_constant(inputs[1])).toTensor(); - Weights w = Weights(w_tensor); - - nvinfer1::ILayer* new_layer; - auto maybe_bias = engine->context().get_constant(inputs[2]); - if (maybe_bias.isTensor()) { - auto bias = maybe_bias.toTensor(); - Weights b = Weights(bias); - new_layer = engine->network()->addFullyConnected(*in, w.outputs_num, w.data, b.data); - } else { - LOG(INFO) << "There is no bias for the linear layer"; - new_layer = engine->network()->addFullyConnected(*in, w.outputs_num, w.data, Weights().data); - } - POROS_CHECK(new_layer, "Unable to create linear layer from node: " << *node); - new_layer->setName((layer_info(node) + "_IFullyConnectedLayer").c_str()); - engine->context().set_tensor(node->outputs()[0], new_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << new_layer->getOutput(0)->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, LinearConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/linear.h b/poros/poros/converter/gpu/linear.h deleted file mode 100644 index 939215443bb..00000000000 --- a/poros/poros/converter/gpu/linear.h +++ /dev/null @@ -1,61 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file linear.h -* @author tianjinjin@baidu.com -* @date Fri Aug 20 17:21:44 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class LinearConverter : public GpuConverter { -public: - LinearConverter() {} - virtual ~LinearConverter() {} - - //当前使用的版本 - //参考了pytorch的实现,根据情况将其转换成addmm或者matmul + add。 - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - //DEPRECATED: 调用addFullyConnected进行组网的版本 - //在transform的模型中遇到了dimention不一致的问题,先搁置。 - bool converter_fully_connect_version(TensorrtEngine* engine, const torch::jit::Node *node); - - virtual const std::vector schema_string() { - return {"aten::linear(Tensor input, Tensor weight, Tensor? bias=None) -> Tensor"}; - } - - virtual const std::vector node_kind() { - return {torch::jit::aten::linear}; - } - -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/converter/gpu/list.cpp b/poros/poros/converter/gpu/list.cpp deleted file mode 100644 index 7b33a8e25a3..00000000000 --- a/poros/poros/converter/gpu/list.cpp +++ /dev/null @@ -1,184 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file list.cpp -* @author tianjinjin@baidu.com -* @date Mon Mar 8 11:36:11 CST 2021 -* @brief -**/ -#include "torch/script.h" - -#include "poros/converter/gpu/converter_util.h" -#include "poros/converter/gpu/list.h" -#include "poros/converter/gpu/weight.h" -#include "poros/context/poros_global.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -bool ListConstructConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - const torch::jit::Value* output = node->outputs()[0]; - const auto num_inputs = inputs.size(); - //typical situation: Construct a TensorList - if (output->type()->isSubtypeOf(c10::ListType::ofTensors()) || - output->type()->str().find("Tensor?[]") != std::string::npos) { - std::vector tensorlist; - tensorlist.reserve(num_inputs); - for (auto in : inputs) { - auto in_tensor = engine->context().get_tensor(in); - POROS_CHECK_TRUE((in_tensor != nullptr), "Unable to extract in_tensor for node: " << *node); - tensorlist.emplace_back(in_tensor); - } - engine->context().set_tensorlist(node->outputs()[0], tensorlist); - - // IntList - } else if (output->type()->isSubtypeOf(c10::ListType::ofInts())) { - // 检查int是否以nvtensor的形式输入 - if (check_inputs_tensor_scalar(engine, node)) { - std::vector inputs_nvtensor; - // 将所有int对应的nvtensor加入vector, 最后cat起来 - for (auto in : inputs) { - nvinfer1::ITensor* temp_tensor = this->get_tensor_scalar(in); - POROS_CHECK_TRUE((temp_tensor != nullptr), node_info(node) + std::string("get int nvtensor false.")); - inputs_nvtensor.push_back(temp_tensor); - } - nvinfer1::IConcatenationLayer* concat_layer = - engine->network()->addConcatenation(inputs_nvtensor.data(), inputs_nvtensor.size()); - // 这里确保输出类型是int - concat_layer->setOutputType(0, nvinfer1::DataType::kINT32); - concat_layer->setName((layer_info(node) + "_IConcatenationLayer").c_str()); - concat_layer->setAxis(0); - engine->context().set_tensor(node->outputs()[0], concat_layer->getOutput(0)); - } - else { - // 输入是正常情况 - c10::List list; - list.reserve(num_inputs); - for (auto in : inputs) { - auto in_const = engine->context().get_constant(in); - list.emplace_back(std::move(in_const.toInt())); - } - auto output_ivalue = c10::optional(std::move(torch::jit::IValue(list))); - engine->context().set_constant(node->outputs()[0], output_ivalue); - } - - // FloatList - } else if (output->type()->isSubtypeOf(c10::ListType::ofFloats())) { - c10::List list; - list.reserve(num_inputs); - for (auto in : inputs) { - auto in_const = engine->context().get_constant(in); - list.emplace_back(std::move(in_const.toDouble())); - } - auto output_ivalue = c10::optional(std::move(torch::jit::IValue(list))); - engine->context().set_constant(node->outputs()[0], output_ivalue); - - // BoolList - } else if (output->type()->isSubtypeOf(c10::ListType::ofBools())) { - c10::List list; - list.reserve(num_inputs); - for (auto in : inputs) { - auto in_const = engine->context().get_constant(in); - list.emplace_back(std::move(in_const.toBool())); - } - auto output_ivalue = c10::optional(std::move(torch::jit::IValue(list))); - engine->context().set_constant(node->outputs()[0], output_ivalue); - - //TODO: meet some unsupported type - } else { - POROS_THROW_ERROR("Meet some unsupported output value type in ListConstructConverter" << *node); - } - return true; -} - -bool ListUnpackConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - at::ArrayRef outputs = node->outputs(); - // 检查int[]是否以nvtensor的形式输入 - if (check_inputs_tensor_scalar(engine, node)) { - nvinfer1::ITensor* input_int_nvtensor = get_tensor_scalar(inputs[0]); - POROS_CHECK_TRUE((input_int_nvtensor != nullptr), node_info(node) + std::string("get int nvtensor false.")); - - nvinfer1::Dims input_dims = input_int_nvtensor->getDimensions(); - // int[]只有一维数据, 获得要unpack的int数量 - int64_t dim_rank = input_dims.d[0]; - POROS_CHECK_TRUE((outputs.size() == (size_t)dim_rank), - "the input list size do not equal output size for ListUnpackConverter as expected"); - // int[] - for (int64_t i = 0; i < dim_rank; i++) { - std::vector start_vec{i}, size_vec{1}, stride_vec{1}; - auto slice_layer = engine->network()->addSlice(*input_int_nvtensor, - sizes_to_nvdim(start_vec), - sizes_to_nvdim(size_vec), - sizes_to_nvdim(stride_vec)); - POROS_CHECK(slice_layer, "Unable to given dim info from node: " << *node); - slice_layer->setName((layer_info(node) + "_ISliceLayer" + std::to_string(i)).c_str()); - nvinfer1::ITensor* slice_output = slice_layer->getOutput(0); - engine->context().set_tensor(outputs[i], slice_output); - } - return true; - } - - std::vector output_vec; - engine->context().get_tensorlist(inputs[0], output_vec); - POROS_CHECK_TRUE((outputs.size() == output_vec.size()), - "the input list size do not equal output size for ListUnpackConverter as expected"); - - //TODO: check if this implement is right, check output is tuple or mulituple ivalues. - for (size_t index = 0; index < outputs.size(); index++) { - auto out = outputs[index]; - //Tensor situation - if (out->type()->isSubtypeOf(c10::TensorType::get())) { - engine->context().set_tensor(out, output_vec[index]); - } else { - POROS_THROW_ERROR("Meet some unsupported output value type in ListUnpackConverter" << *node); - } - } - return true; -} - -// OP aten::list, original python code looks like: "for a_shape in list(data.shape): ......" -bool ListConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - const torch::jit::Value* output = node->outputs()[0]; - const auto num_inputs = inputs.size(); - POROS_CHECK_TRUE((num_inputs == 1),"More than 1 input is not supported for node:" << *node) - auto input = inputs[0]; - POROS_CHECK_TRUE((input->type()->str() == output->type()->str()),"Input and Output are in different types") - auto input_tensor = engine->context().get_tensor(input); - if (!input_tensor) { - std::vector tensor_list; - POROS_CHECK_TRUE(engine->context().get_tensorlist(input, tensor_list), "extract tensor list err"); - engine->context().set_tensorlist(node->outputs()[0], tensor_list); - } - else { - engine->context().set_tensor(node->outputs()[0],input_tensor); - } - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, ListConstructConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, ListUnpackConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, ListConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/list.h b/poros/poros/converter/gpu/list.h deleted file mode 100644 index 12f4555e382..00000000000 --- a/poros/poros/converter/gpu/list.h +++ /dev/null @@ -1,89 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file list.h -* @author tianjinjin@baidu.com -* @date Tue Jul 27 11:24:21 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class ListConstructConverter : public GpuConverter { -public: - ListConstructConverter() {} - virtual ~ListConstructConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - //prim::ListConstruct kind node has no schema - const std::vector schema_string() { - return {}; - } - - const std::vector node_kind() { - return {torch::jit::prim::ListConstruct}; - } -}; - -class ListUnpackConverter : public GpuConverter { -public: - ListUnpackConverter() {} - virtual ~ListUnpackConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - //prim::ListUnpack kind node has no schema - const std::vector schema_string() { - return {}; - } - - const std::vector node_kind() { - return {torch::jit::prim::ListUnpack}; - } -}; - -class ListConverter : public GpuConverter { -public: - ListConverter() {} - virtual ~ListConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - //prim::List kind node has no schema - const std::vector schema_string() { - return {}; - } - - const std::vector node_kind() { - return {torch::jit::aten::list}; - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/logical.cpp b/poros/poros/converter/gpu/logical.cpp deleted file mode 100644 index bf952202624..00000000000 --- a/poros/poros/converter/gpu/logical.cpp +++ /dev/null @@ -1,162 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file logical.cpp -* @author Lin Xiao Chun (linxiaochun@baidu.com) -* @date 2022-02-17 18:32:04 -* @brief -**/ - -#include "poros/converter/gpu/logical.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -bool AndConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for AndConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for AndConverter is not Tensor as expected"); - POROS_CHECK_TRUE(((inputs[0]->node()->kind() != torch::jit::prim::Constant) && - (inputs[1]->node()->kind() != torch::jit::prim::Constant)), - "constant input is not support for AndConverter"); - - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - - auto other = engine->context().get_tensor(inputs[1]); - POROS_CHECK_TRUE((other != nullptr), "Unable to init input tensor for node: " << *node); - - POROS_CHECK_TRUE(((self->getType() == nvinfer1::DataType::kBOOL) && (other->getType() == nvinfer1::DataType::kBOOL)), - "Only Bool type supported for for node: " << *node); - - POROS_CHECK_TRUE(((self->getDimensions().nbDims > 0) && (other->getDimensions().nbDims > 0)), - "scalar input is not supported for node: " << *node); - - auto new_layer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kAND, - self, - other, - layer_info(node) + "_and"); - - POROS_CHECK(new_layer, "Unable to create And layer from node: " << *node); - engine->context().set_tensor(node->outputs()[0], new_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << new_layer->getOutput(0)->getDimensions(); - return true; -} - - -bool OrConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for OrConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for OrConverter is not Tensor as expected"); - POROS_CHECK_TRUE(((inputs[0]->node()->kind() != torch::jit::prim::Constant) && - (inputs[1]->node()->kind() != torch::jit::prim::Constant)), - "constant input is not support for OrConverter"); - - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - - auto other = engine->context().get_tensor(inputs[1]); - POROS_CHECK_TRUE((other != nullptr), "Unable to init input tensor for node: " << *node); - - POROS_CHECK_TRUE(((self->getType() == nvinfer1::DataType::kBOOL) && (other->getType() == nvinfer1::DataType::kBOOL)), - "Only Bool type supported for for node: " << *node); - - POROS_CHECK_TRUE(((self->getDimensions().nbDims > 0) && (other->getDimensions().nbDims > 0)), - "scalar input is not supported for node: " << *node); - - auto new_layer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kOR, - self, - other, - layer_info(node) + "_or"); - - POROS_CHECK(new_layer, "Unable to create Or layer from node: " << *node); - engine->context().set_tensor(node->outputs()[0], new_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << new_layer->getOutput(0)->getDimensions(); - return true; -} - -bool XorConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for XorConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for XorConverter is not Tensor as expected"); - POROS_CHECK_TRUE(((inputs[0]->node()->kind() != torch::jit::prim::Constant) && - (inputs[1]->node()->kind() != torch::jit::prim::Constant)), - "constant input is not support for XorConverter"); - - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - - auto other = engine->context().get_tensor(inputs[1]); - POROS_CHECK_TRUE((other != nullptr), "Unable to init input tensor for node: " << *node); - - POROS_CHECK_TRUE(((self->getType() == nvinfer1::DataType::kBOOL) && (other->getType() == nvinfer1::DataType::kBOOL)), - "Only Bool type supported for for node: " << *node); - - POROS_CHECK_TRUE(((self->getDimensions().nbDims > 0) && (other->getDimensions().nbDims > 0)), - "scalar input is not supported for node: " << *node); - - auto new_layer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kXOR, - self, - other, - layer_info(node) + "_xor"); - - POROS_CHECK(new_layer, "Unable to create Xor layer from node: " << *node); - engine->context().set_tensor(node->outputs()[0], new_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << new_layer->getOutput(0)->getDimensions(); - return true; -} - -bool NotConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 1), "invaid inputs size for NotConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for NotConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[0]->node()->kind() != torch::jit::prim::Constant), - "constant input is not support for NotConverter"); - - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - auto new_layer = engine->network()->addUnary(*self,nvinfer1::UnaryOperation::kNOT); - - POROS_CHECK(new_layer, "Unable to create And layer from node: " << *node); - new_layer->setName((layer_info(node) + "_IUnaryLayer").c_str()); - - engine->context().set_tensor(node->outputs()[0], new_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << new_layer->getOutput(0)->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, AndConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, OrConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, XorConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, NotConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/logical.h b/poros/poros/converter/gpu/logical.h deleted file mode 100644 index b168d6a50ad..00000000000 --- a/poros/poros/converter/gpu/logical.h +++ /dev/null @@ -1,139 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file logical.h -* @author Lin Xiao Chun (linxiaochun@baidu.com) -* @date 2022-02-17 18:32:23 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class AndConverter : public GpuConverter { -public: - AndConverter() {} - virtual ~AndConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - //aten::__and__.Scalar(Tensor self, Scalar other) -> Tensor - const std::vector schema_string() { - return { - "aten::__and__.Tensor(Tensor self, Tensor other) -> Tensor", - "aten::bitwise_and.Tensor(Tensor self, Tensor other) -> Tensor", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * - * **/ - const std::vector node_kind() { - return {torch::jit::aten::__and__, - torch::jit::aten::__iand__, - torch::jit::aten::bitwise_and, - }; - } -}; - -class OrConverter : public GpuConverter { -public: - OrConverter() {} - virtual ~OrConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return { - "aten::__or__.Tensor(Tensor self, Tensor other) -> Tensor", - "aten::bitwise_or.Tensor(Tensor self, Tensor other) -> Tensor", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * - * **/ - const std::vector node_kind() { - return {torch::jit::aten::__or__, - torch::jit::aten::__ior__, - torch::jit::aten::bitwise_or, - }; - } -}; - -class XorConverter : public GpuConverter { -public: - XorConverter() {} - virtual ~XorConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return { - "aten::__xor__.Tensor(Tensor self, Tensor other) -> Tensor", - "aten::bitwise_xor.Tensor(Tensor self, Tensor other) -> Tensor", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * - * **/ - const std::vector node_kind() { - return {torch::jit::aten::__xor__, - torch::jit::aten::__ixor__, - torch::jit::aten::bitwise_xor, - }; - } -}; - -class NotConverter : public GpuConverter { -public: - NotConverter() {} - virtual ~NotConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - //aten::bitwise_not(Tensor self) -> Tensor - const std::vector schema_string() { - return { - "aten::bitwise_not(Tensor self) -> Tensor", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * - * **/ - const std::vector node_kind() { - return { - torch::jit::aten::bitwise_not, - }; - - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/lstm.cpp b/poros/poros/converter/gpu/lstm.cpp deleted file mode 100644 index 8d59998e24b..00000000000 --- a/poros/poros/converter/gpu/lstm.cpp +++ /dev/null @@ -1,195 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file lstm.cpp -* @author wangrui39@baidu.com -* @date Mon December 13 11:36:11 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/lstm.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" -#include "poros/converter/gpu/add.h" - -namespace baidu { -namespace mirana { -namespace poros { - -bool add_rnnv2_params(at::Tensor params, nvinfer1::IRNNv2Layer* &layer, bool isW, int rela_index, - int hidden_size, int idx, nvinfer1::RNNGateType gate, bool bias = false) { - std::vector w; - for (int i = idx * hidden_size; i < hidden_size * (idx + 1); i++){ - w.push_back(params[i].unsqueeze(0)); - } - if (bias) { - layer->setBiasForGate(rela_index, gate, isW, Weights(at::cat(w, 0)).data); - } - else { - layer->setWeightsForGate(rela_index, gate, isW, Weights(at::cat(w, 0)).data); - } - return true; -} - -bool LstmConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - /*aten::lstm.input(Tensor input, - Tensor[] hx, - Tensor[] params, - bool has_biases, - int num_layers, - float dropout, - bool train, - bool bidirectional, - bool batch_first) -> (Tensor, Tensor, Tensor))*/ - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 9), "invaid inputs size for LstmConverter"); - - // 获取输入 - nvinfer1::ITensor *input = engine->context().get_tensor(inputs[0]); - std::vector hx_tensorlist; - engine->context().get_tensorlist(inputs[1], hx_tensorlist); - POROS_CHECK_TRUE((hx_tensorlist.size() == 2), "Unable to init input List[tensor] for node: " << *node); - - // 获取参数 - torch::jit::IValue params = engine->context().get_constant(inputs[2]); - POROS_CHECK_TRUE((params.isTensorList()), "Unable to init second input List[tensor] for node: " << *node); - c10::List param_list = params.toTensorList(); - int num_layers = engine->context().get_constant(inputs[4]).toInt(); - bool bidirectional = engine->context().get_constant(inputs[7]).toBool(); - bool batch_first = engine->context().get_constant(inputs[8]).toBool(); - - // 获取构建trt rnnlayer的输入 - nvinfer1::ITensor *h_0 = hx_tensorlist[0]; - nvinfer1::ITensor *c_0 = hx_tensorlist[1]; - int32_t hidden_size = c_0->getDimensions().d[c_0->getDimensions().nbDims - 1]; - - if (!batch_first) { - auto input_shuffle_layer = engine->network()->addShuffle(*input); - input_shuffle_layer->setFirstTranspose(nvinfer1::Permutation{1, 0, 2}); - input_shuffle_layer->setName((layer_info(node) + "_IShuffleLayer_for_input").c_str()); - input = input_shuffle_layer->getOutput(0); - } - int max_seqlen = input->getDimensions().d[1]; - - // 使用trt现有的lstm - auto rnnv2_layer = engine->network()->addRNNv2(*input, num_layers, hidden_size, max_seqlen, nvinfer1::RNNOperation::kLSTM); - if (bidirectional) { - rnnv2_layer->setDirection(nvinfer1::RNNDirection::kBIDIRECTION); - } - rnnv2_layer->setName((layer_info(node) + "_IRNNv2Layer").c_str()); - - auto c_0_shuffle_layer = engine->network()->addShuffle(*c_0); - c_0_shuffle_layer->setFirstTranspose(nvinfer1::Permutation{1, 0, 2}); - c_0_shuffle_layer->setName((layer_info(node) + "_IShuffleLayer_for_c0").c_str()); - rnnv2_layer->setCellState(*c_0_shuffle_layer->getOutput(0)); - - auto h_0_shuffle_layer = engine->network()->addShuffle(*h_0); - h_0_shuffle_layer->setFirstTranspose(nvinfer1::Permutation{1, 0, 2}); - h_0_shuffle_layer->setName((layer_info(node) + "_IShuffleLayer_for_h0").c_str()); - rnnv2_layer->setHiddenState(*h_0_shuffle_layer->getOutput(0)); - - // 循环生成layer - for (int i = 0; i < num_layers; i++){ - size_t rela_index = 0; - if (bidirectional) { - rela_index = 2 * i; - } - else { - rela_index = i; - } - - // weight_ih_l - add_rnnv2_params(param_list[rela_index * 4 + 0], rnnv2_layer, true, rela_index, hidden_size, 0, nvinfer1::RNNGateType::kINPUT); - add_rnnv2_params(param_list[rela_index * 4 + 0], rnnv2_layer, true, rela_index, hidden_size, 1, nvinfer1::RNNGateType::kFORGET); - add_rnnv2_params(param_list[rela_index * 4 + 0], rnnv2_layer, true, rela_index, hidden_size, 2, nvinfer1::RNNGateType::kCELL); - add_rnnv2_params(param_list[rela_index * 4 + 0], rnnv2_layer, true, rela_index, hidden_size, 3, nvinfer1::RNNGateType::kOUTPUT); - - // weight_hh_l - add_rnnv2_params(param_list[rela_index * 4 + 1], rnnv2_layer, false, rela_index, hidden_size, 0, nvinfer1::RNNGateType::kINPUT); - add_rnnv2_params(param_list[rela_index * 4 + 1], rnnv2_layer, false, rela_index, hidden_size, 1, nvinfer1::RNNGateType::kFORGET); - add_rnnv2_params(param_list[rela_index * 4 + 1], rnnv2_layer, false, rela_index, hidden_size, 2, nvinfer1::RNNGateType::kCELL); - add_rnnv2_params(param_list[rela_index * 4 + 1], rnnv2_layer, false, rela_index, hidden_size, 3, nvinfer1::RNNGateType::kOUTPUT); - - // bias_ih_l - add_rnnv2_params(param_list[rela_index * 4 + 2], rnnv2_layer, true, rela_index, hidden_size, 0, nvinfer1::RNNGateType::kINPUT, true); - add_rnnv2_params(param_list[rela_index * 4 + 2], rnnv2_layer, true, rela_index, hidden_size, 1, nvinfer1::RNNGateType::kFORGET, true); - add_rnnv2_params(param_list[rela_index * 4 + 2], rnnv2_layer, true, rela_index, hidden_size, 2, nvinfer1::RNNGateType::kCELL, true); - add_rnnv2_params(param_list[rela_index * 4 + 2], rnnv2_layer, true, rela_index, hidden_size, 3, nvinfer1::RNNGateType::kOUTPUT, true); - - // bias_hh_l - add_rnnv2_params(param_list[rela_index * 4 + 3], rnnv2_layer, false, rela_index, hidden_size, 0, nvinfer1::RNNGateType::kINPUT, true); - add_rnnv2_params(param_list[rela_index * 4 + 3], rnnv2_layer, false, rela_index, hidden_size, 1, nvinfer1::RNNGateType::kFORGET, true); - add_rnnv2_params(param_list[rela_index * 4 + 3], rnnv2_layer, false, rela_index, hidden_size, 2, nvinfer1::RNNGateType::kCELL, true); - add_rnnv2_params(param_list[rela_index * 4 + 3], rnnv2_layer, false, rela_index, hidden_size, 3, nvinfer1::RNNGateType::kOUTPUT, true); - - if (bidirectional) { - // ================reverse===================== - // weight_ih_l - add_rnnv2_params(param_list[(rela_index + 1) * 4 + 0], rnnv2_layer, true, rela_index + 1, hidden_size, 0, nvinfer1::RNNGateType::kINPUT); - add_rnnv2_params(param_list[(rela_index + 1) * 4 + 0], rnnv2_layer, true, rela_index + 1, hidden_size, 1, nvinfer1::RNNGateType::kFORGET); - add_rnnv2_params(param_list[(rela_index + 1) * 4 + 0], rnnv2_layer, true, rela_index + 1, hidden_size, 2, nvinfer1::RNNGateType::kCELL); - add_rnnv2_params(param_list[(rela_index + 1) * 4 + 0], rnnv2_layer, true, rela_index + 1, hidden_size, 3, nvinfer1::RNNGateType::kOUTPUT); - - // weight_hh_l - add_rnnv2_params(param_list[(rela_index + 1) * 4 + 1], rnnv2_layer, false, rela_index + 1, hidden_size, 0, nvinfer1::RNNGateType::kINPUT); - add_rnnv2_params(param_list[(rela_index + 1) * 4 + 1], rnnv2_layer, false, rela_index + 1, hidden_size, 1, nvinfer1::RNNGateType::kFORGET); - add_rnnv2_params(param_list[(rela_index + 1) * 4 + 1], rnnv2_layer, false, rela_index + 1, hidden_size, 2, nvinfer1::RNNGateType::kCELL); - add_rnnv2_params(param_list[(rela_index + 1) * 4 + 1], rnnv2_layer, false, rela_index + 1, hidden_size, 3, nvinfer1::RNNGateType::kOUTPUT); - - // bias_ih_l - add_rnnv2_params(param_list[(rela_index + 1) * 4 + 2], rnnv2_layer, true, rela_index + 1, hidden_size, 0, nvinfer1::RNNGateType::kINPUT, true); - add_rnnv2_params(param_list[(rela_index + 1) * 4 + 2], rnnv2_layer, true, rela_index + 1, hidden_size, 1, nvinfer1::RNNGateType::kFORGET, true); - add_rnnv2_params(param_list[(rela_index + 1) * 4 + 2], rnnv2_layer, true, rela_index + 1, hidden_size, 2, nvinfer1::RNNGateType::kCELL, true); - add_rnnv2_params(param_list[(rela_index + 1) * 4 + 2], rnnv2_layer, true, rela_index + 1, hidden_size, 3, nvinfer1::RNNGateType::kOUTPUT, true); - - // bias_hh_l - add_rnnv2_params(param_list[(rela_index + 1) * 4 + 3], rnnv2_layer, false, rela_index + 1, hidden_size, 0, nvinfer1::RNNGateType::kINPUT, true); - add_rnnv2_params(param_list[(rela_index + 1) * 4 + 3], rnnv2_layer, false, rela_index + 1, hidden_size, 1, nvinfer1::RNNGateType::kFORGET, true); - add_rnnv2_params(param_list[(rela_index + 1) * 4 + 3], rnnv2_layer, false, rela_index + 1, hidden_size, 2, nvinfer1::RNNGateType::kCELL, true); - add_rnnv2_params(param_list[(rela_index + 1) * 4 + 3], rnnv2_layer, false, rela_index + 1, hidden_size, 3, nvinfer1::RNNGateType::kOUTPUT, true); - } - } - - nvinfer1::ITensor* output = rnnv2_layer->getOutput(0); - if (!batch_first) { - auto output1_shuffle_layer = engine->network()->addShuffle(*output); - output1_shuffle_layer->setFirstTranspose(nvinfer1::Permutation{1, 0, 2}); - output1_shuffle_layer->setName((layer_info(node) + "_IShuffleLayer_for_output").c_str()); - output = output1_shuffle_layer->getOutput(0); - } - auto output2_shuffle_layer = engine->network()->addShuffle(*rnnv2_layer->getOutput(1)); - output2_shuffle_layer->setFirstTranspose(nvinfer1::Permutation{1, 0, 2}); - output2_shuffle_layer->setName((layer_info(node) + "_IShuffleLayer_for_output1").c_str()); - auto output3_shuffle_layer = engine->network()->addShuffle(*rnnv2_layer->getOutput(2)); - output3_shuffle_layer->setFirstTranspose(nvinfer1::Permutation{1, 0, 2}); - output3_shuffle_layer->setName((layer_info(node) + "_IShuffleLayer_for_output2").c_str()); - - engine->context().set_tensor(node->outputs()[0], output); - engine->context().set_tensor(node->outputs()[1], output2_shuffle_layer->getOutput(0)); - engine->context().set_tensor(node->outputs()[2], output3_shuffle_layer->getOutput(0)); - - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, LstmConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/lstm.h b/poros/poros/converter/gpu/lstm.h deleted file mode 100644 index eb69ffd127d..00000000000 --- a/poros/poros/converter/gpu/lstm.h +++ /dev/null @@ -1,63 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file lstm.h -* @author wangrui39@baidu.com -* @date Mon January 17 11:36:11 CST 2022 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -// Correspons to torch.lstm_cell https://pytorch.org/docs/1.9.1/generated/torch.nn.LSTM.html?highlight=lstm#torch.nn.LSTM -class LstmConverter : public GpuConverter { -public: - LstmConverter() {} - virtual ~LstmConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::lstm.input(Tensor input, Tensor[] hx, Tensor[] params, bool has_biases, int num_layers, float dropout, bool train, bool bidirectional, bool batch_first) -> (Tensor, Tensor, Tensor)"}; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * aten::lstm.input(Tensor input, - * Tensor[] hx, Tensor[] params, - * bool has_biases, int num_layers, float dropout, - * bool train, - * bool bidirectional, - * bool batch_first) -> (Tensor, Tensor, Tensor)) - * **/ - const std::vector node_kind() { - return {torch::jit::aten::lstm}; - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/lstm_cell.cpp b/poros/poros/converter/gpu/lstm_cell.cpp deleted file mode 100644 index 3eb12685e18..00000000000 --- a/poros/poros/converter/gpu/lstm_cell.cpp +++ /dev/null @@ -1,197 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file lstm_cell.cpp -* @author wangrui39@baidu.com -* @date Mon December 13 11:36:11 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/lstm_cell.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" -#include "poros/converter/gpu/add.h" - -namespace baidu { -namespace mirana { -namespace poros { - -nvinfer1::Dims todims_pad(nvinfer1::Dims s_dim, int32_t pad_to) { - if (s_dim.nbDims > pad_to){ - LOG(WARNING) << "Requested padding of dimensions to " << pad_to << " but found " << - s_dim.nbDims << " dimensions, not going to pad"; - return s_dim; - } - - nvinfer1::Dims dims; - dims.nbDims = pad_to; - for (int32_t i = 0; i < pad_to - s_dim.nbDims; ++i) { - dims.d[i] = 1; - } - for (int32_t i = pad_to - s_dim.nbDims; i < pad_to; ++i) { - dims.d[i] = s_dim.d[i - (pad_to - s_dim.nbDims)]; - } - return dims; -} - -nvinfer1::ITensor* calculate_gate( - TensorrtEngine* engine, nvinfer1::ITensor *input, nvinfer1::ITensor *w, - std::string b_name = "", nvinfer1::ITensor *b = nullptr) { - - auto mm = engine->network()->addMatrixMultiply( - *input, nvinfer1::MatrixOperation::kNONE, *w, nvinfer1::MatrixOperation::kTRANSPOSE); - nvinfer1::ITensor *mm_out = mm->getOutput(0); - - if (b != nullptr) { - auto mout_dim = mm_out->getDimensions(); - auto b_dim = b->getDimensions(); - - if (mout_dim.d != b_dim.d) { - auto shuffle = engine->network()->addShuffle(*b); - shuffle->setReshapeDimensions(todims_pad(b_dim, mout_dim.nbDims)); - b = shuffle->getOutput(0); - } - - auto add_layer = engine->network()->addElementWise(*mm_out, *b, nvinfer1::ElementWiseOperation::kSUM); - return add_layer->getOutput(0); - } - else{ - return mm_out; - } - -} - -bool LstmCellConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - //aten::lstm_cell(Tensor input, Tensor[] hx, Tensor w_ih, Tensor w_hh, Tensor? b_ih=None, Tensor? b_hh=None) -> (Tensor, Tensor) - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "inputs[0] for LstmCellConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[1]->type()->isSubtypeOf(c10::ListType::ofTensors())), - "inputs[1] for LstmCellConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[2]->type()->isSubtypeOf(c10::TensorType::get())), - "inputs[2] for LstmCellConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[3]->type()->isSubtypeOf(c10::TensorType::get())), - "inputs[3] for LstmCellConverter is not Tensor as expected"); - - //extract Tensors[] - std::vector state; - bool ret = engine->context().get_tensorlist(inputs[1], state); - POROS_CHECK_TRUE((state.size() == 2), "Unable to init input List[tensor] for node: " << *node); - POROS_CHECK_TRUE(ret, "Unable to init input List[tensor] for node: " << *node); - - //extract Tensor - nvinfer1::ITensor *input = engine->context().get_tensor(inputs[0]); - nvinfer1::ITensor *w_ih = engine->context().get_tensor(inputs[2]); - nvinfer1::ITensor *w_hh = engine->context().get_tensor(inputs[3]); - - // calculate first half of gates - nvinfer1::ITensor *out1 = nullptr; - nvinfer1::ITensor *out2 = nullptr; - - if (inputs[4]->type()->isSubtypeOf(c10::TensorType::get())) { - nvinfer1::ITensor *b_ih = engine->context().get_tensor(inputs[4]); - out1 = calculate_gate(engine, input, w_ih, "b_ih", b_ih); - } - else { - out1 = calculate_gate(engine, input, w_ih); - } - POROS_CHECK_TRUE((out1 != nullptr), "invaid b_ih size for ConcatConverter"); - - // calculate second half of gates - if (inputs[5]->type()->isSubtypeOf(c10::TensorType::get())) { - nvinfer1::ITensor *b_hh = engine->context().get_tensor(inputs[5]); - out2 = calculate_gate(engine, state[0], w_hh, "b_hh", b_hh); - } - else { - out2 = calculate_gate(engine, state[0], w_hh); - } - POROS_CHECK_TRUE((out2 != nullptr), "invaid b_hh size for ConcatConverter"); - - // get all 4 gates - auto add_layer = engine->network()->addElementWise(*out1, *out2, nvinfer1::ElementWiseOperation::kSUM); - add_layer->setName((layer_info(node) + "_sum_" + "for_add_out").c_str()); - nvinfer1::ITensor *add_out = add_layer->getOutput(0); - - // chunk Tensor into 4 parts and apply activation functions - auto dims = add_out->getDimensions().d; - auto batch = dims[0]; - auto hidden = dims[1] / 4; - - auto size = nvinfer1::Dims2(batch, hidden); - auto stride = nvinfer1::Dims2(1, 1); - auto offset0 = nvinfer1::Dims2(0, 0); - auto offset1 = nvinfer1::Dims2(0, hidden); - auto offset2 = nvinfer1::Dims2(0, 2 * hidden); - auto offset3 = nvinfer1::Dims2(0, 3 * hidden); - - auto slice1 = engine->network()->addSlice(*add_out, offset0, size, stride); - slice1->setName((layer_info(node) + "_ISliceLayer_" + "for_offset0").c_str()); - auto active1 = engine->network()->addActivation(*slice1->getOutput(0), nvinfer1::ActivationType::kSIGMOID); - active1->setName((layer_info(node) + "_IActivationLayer_" + "for_offset0").c_str()); - auto ingate = active1->getOutput(0); - - auto slice2 = engine->network()->addSlice(*add_out, offset1, size, stride); - slice2->setName((layer_info(node) + "_ISliceLayer_" + "for_offset1").c_str()); - auto active2 = engine->network()->addActivation(*slice2->getOutput(0), nvinfer1::ActivationType::kSIGMOID); - active2->setName((layer_info(node) + "_IActivationLayer_" + "for_offset1").c_str()); - auto forgetgate = active2->getOutput(0); - - auto slice3 = engine->network()->addSlice(*add_out, offset2, size, stride); - slice3->setName((layer_info(node) + "_ISliceLayer_" + "for_offset2").c_str()); - auto active3 = engine->network()->addActivation(*slice3->getOutput(0), nvinfer1::ActivationType::kTANH); - active3->setName((layer_info(node) + "_IActivationLayer_" + "for_offset2").c_str()); - auto cellgate = active3->getOutput(0); - - auto slice4 = engine->network()->addSlice(*add_out, offset3, size, stride); - slice4->setName((layer_info(node) + "_ISliceLayer_" + "for_offset3").c_str()); - auto active4 = engine->network()->addActivation(*slice4->getOutput(0), nvinfer1::ActivationType::kSIGMOID); - active4->setName((layer_info(node) + "_IActivationLayer_" + "for_offset3").c_str()); - auto outgate = active4->getOutput(0); - - // compute cy - auto forget_cx = engine->network()->addElementWise(*forgetgate, *state[1], nvinfer1::ElementWiseOperation::kPROD); - forget_cx->setName((layer_info(node) + "_prod_" + "for_forget_cx").c_str()); - auto in_cell = engine->network()->addElementWise(*ingate, *cellgate, nvinfer1::ElementWiseOperation::kPROD); - in_cell->setName((layer_info(node) + "_prod_" + "for_in_cell").c_str()); - auto cy = engine->network()->addElementWise( - *forget_cx->getOutput(0), *in_cell->getOutput(0), nvinfer1::ElementWiseOperation::kSUM); - cy->setName((layer_info(node) + "_prod_" + "for_cy").c_str()); - auto cy_out = cy->getOutput(0); - - // compute hy - auto cy_tanh = engine->network()->addActivation(*cy_out, nvinfer1::ActivationType::kTANH); - cy_tanh->setName((layer_info(node) + "_IActivationLayer_" + "for_cy_tanh").c_str()); - auto hy = engine->network()->addElementWise(*outgate, *cy_tanh->getOutput(0), nvinfer1::ElementWiseOperation::kPROD); - hy->setName((layer_info(node) + "_prod_" + "for_hy").c_str()); - auto hy_out = hy->getOutput(0); - - engine->context().set_tensor(node->outputs()[0], hy_out); - engine->context().set_tensor(node->outputs()[1], cy_out); - - LOG(INFO) << "Output tensor shape: " << hy_out->getDimensions(); - LOG(INFO) << "Output tensor shape: " << cy_out->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, LstmCellConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/lstm_cell.h b/poros/poros/converter/gpu/lstm_cell.h deleted file mode 100644 index 17d2c139740..00000000000 --- a/poros/poros/converter/gpu/lstm_cell.h +++ /dev/null @@ -1,63 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file lstm_cell.h -* @author wangrui39@baidu.com -* @date Mon December 13 11:36:11 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -// Correspons to torch.lstm_cell https://pytorch.org/docs/stable/generated/torch.nn.LSTMCell.htmls -class LstmCellConverter : public GpuConverter { -public: - LstmCellConverter() {} - virtual ~LstmCellConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - //aten::lstm_cell(Tensor input, Tensor[] hx, Tensor w_ih, Tensor w_hh, Tensor? b_ih=None, Tensor? b_hh=None) -> (Tensor, Tensor) - const std::vector schema_string() { - return {"aten::lstm_cell(Tensor input, Tensor[] hx, Tensor w_ih, Tensor w_hh, Tensor? b_ih=None, Tensor? b_hh=None) -> (Tensor, Tensor)"}; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::lstm_cell(Tensor input, Tensor[] hx, Tensor w_ih, Tensor w_hh, Tensor? b_ih=None, Tensor? b_hh=None) -> (Tensor, Tensor)", - * **/ - const std::vector node_kind() { - return {torch::jit::aten::lstm_cell}; - } - - bool assign_schema_attr() { - return assign_schema_attr_helper({{"aten::lstm_cell(Tensor input, Tensor[] hx, Tensor w_ih, Tensor w_hh, Tensor? b_ih=None, Tensor? b_hh=None) -> (Tensor, Tensor)", {0, 0}}}); - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/matrix_multiply.cpp b/poros/poros/converter/gpu/matrix_multiply.cpp deleted file mode 100644 index f892a3fdb88..00000000000 --- a/poros/poros/converter/gpu/matrix_multiply.cpp +++ /dev/null @@ -1,421 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file matrix_multiply.cpp -* @author tianjinjin@baidu.com -* @date Mon Mar 8 11:36:11 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/converter_util.h" -#include "poros/converter/gpu/matrix_multiply.h" -#include "poros/converter/gpu/weight.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/** aten::matmul(Tensor self, Tensor other) -> Tensor - * this can be much complicated than i expected. - * The behavior depends on the dimensionality of the Tensors as follows: -- If both Tensors are 1-dimensional, the dot product (scalar) is returned. -- If both arguments are 2-dimensional, the matrix-matrix product is returned. -- If the first argument is 1-dimensional and the second argument is 2-dimensional, - a 1 is prepended to its dimension for the purpose of the matrix multiply. - After the matrix multiply, the prepended dimension is removed. -- If the first argument is 2-dimensional and the second argument is 1-dimensional, - the matrix-vector product is returned. -- If both arguments are at least 1-dimensional and at least one argument is - N-dimensional (where N > 2), then a batched matrix multiply is returned. If the first - argument is 1-dimensional, a 1 is prepended to its dimension for the purpose of the - batched matrix multiply and removed after. If the second argument is 1-dimensional, a - 1 is appended to its dimension for the purpose of the batched matrix multiple and removed after. - The non-matrix (i.e. batch) dimensions are broadcasted (and thus - must be broadcastable). For example, if tensor1 is a (j x 1 x n x m) Tensor - and tensor2 is a (k x m x p) Tensor, the returned tensor will be an (j x k x n x p) Tensor. -the original pytorch implementation is: -https://github.com/pytorch/pytorch/blob/v1.9.0/aten/src/ATen/native/LinearAlgebra.cpp#L1354 -*/ -nvinfer1::ITensor* MatmulConverter::converter(TensorrtEngine* engine, - const torch::jit::Node *node, - nvinfer1::ITensor* self, - nvinfer1::ITensor* other) { - auto self_dim = self->getDimensions().nbDims; - auto other_dim = other->getDimensions().nbDims; - auto origin_self_size = nvdim_to_sizes(self->getDimensions()); - auto origin_other_size = nvdim_to_sizes(other->getDimensions()); - - LOG(INFO) << "self dim info : " << self->getDimensions() << " and other dim info: " << other->getDimensions(); - - nvinfer1::ILayer* mm_layer = nullptr; - //situation one: both tensors are 1D. this is like aten::dot - if (self_dim == 1 && other_dim == 1) { - mm_layer = engine->network()->addMatrixMultiply( - *self, nvinfer1::MatrixOperation::kVECTOR, *other, nvinfer1::MatrixOperation::kVECTOR); - - //situation two: input tensor is 1D. - } else if (self_dim == 1 && other_dim == 2) { - mm_layer = engine->network()->addMatrixMultiply( - *self, nvinfer1::MatrixOperation::kVECTOR, *other, nvinfer1::MatrixOperation::kNONE); - - //situation three: other tensor is 1D. - } else if (self_dim == 2 && other_dim == 1) { - mm_layer = engine->network()->addMatrixMultiply( - *self, nvinfer1::MatrixOperation::kNONE, *other, nvinfer1::MatrixOperation::kVECTOR); - - //situation four: input tensor is N-D(N > 2) and other tensor is 1D or 2D - } else if (self_dim > 2 && (other_dim == 1 || other_dim == 2)) { - if (other_dim == 1) { - auto other_shuffle = engine->network()->addShuffle(*other); - POROS_CHECK(other_shuffle, "Unable to create other shuffle layer for MatmulConverter"); - other_shuffle->setReshapeDimensions(unsqueeze_dims(other->getDimensions(), 1)); - other_shuffle->setName((layer_info(node) + "_IShuffleLayer_for_other").c_str()); - other = other_shuffle->getOutput(0); - LOG(INFO) << "after shuffle other dim info turn to: " << other->getDimensions(); - } - - //prepare output_size info - std::vector output_size; - output_size.insert(output_size.end(), origin_self_size.begin(), origin_self_size.end() - 1); - if (other_dim == 2) { - auto other_size = nvdim_to_sizes(other->getDimensions()); - output_size.push_back(other_size[1]); - } - - std::vector new_order = {-1, origin_self_size[self_dim -1]}; - auto self_shuffle = engine->network()->addShuffle(*self); - POROS_CHECK(self_shuffle, "Unable to create self shuffle layer for MatmulConverter"); - self_shuffle->setReshapeDimensions(sizes_to_nvdim(new_order)); - self_shuffle->setName((layer_info(node) + "_IShuffleLayer_for_self").c_str()); - self = self_shuffle->getOutput(0); - LOG(INFO) << "after shuffle self dim info turn to: " << self->getDimensions(); - - auto tmp_mm_layer = engine->network()->addMatrixMultiply( - *self, nvinfer1::MatrixOperation::kNONE, *other, nvinfer1::MatrixOperation::kNONE); - POROS_CHECK(tmp_mm_layer, "Unable to create matrixmul layer for MatmulConverter"); - tmp_mm_layer->setName((layer_info(node) + "_IMatrixMultiplyLayer").c_str()); - auto tmp_output = tmp_mm_layer->getOutput(0); - LOG(INFO) << "matmul output dim info : " << tmp_output->getDimensions(); - - auto out_shuffle = engine->network()->addShuffle(*tmp_output); - POROS_CHECK(out_shuffle, "Unable to create shuffle layer for MatmulConverter"); - out_shuffle->setReshapeDimensions(sizes_to_nvdim(output_size)); - self_shuffle->setName((layer_info(node) + "_IShuffleLayer_for_out").c_str()); - auto output = out_shuffle->getOutput(0); - LOG(INFO) << "reshape output back to original dim info : " << tmp_output->getDimensions(); - return output; - - //situation five: input tensor is N-D(N > 2) and other tensor is 1D or 2D - } else if (other_dim > 2 && (self_dim == 1 || self_dim == 2)) { - const int64_t n = self_dim == 2 ? origin_self_size[0] : 1; - const int64_t m = origin_self_size[self_dim - 1]; - const int64_t p = origin_other_size[other_dim - 1]; - - //let's do other.transpose(-1, -2) - std::vector new_order; - for (int i = 0; i < other_dim; i++) { - new_order.push_back(i); - } - new_order[other_dim - 1] = new_order[other_dim - 2]; - new_order[other_dim - 2] = other_dim - 1; - auto other_shuffle = engine->network()->addShuffle(*other); - POROS_CHECK(other_shuffle, "Unable to create shuffle layer from node: " << *node); - nvinfer1::Permutation permute; - std::copy(new_order.begin(), new_order.end(), permute.order); - other_shuffle->setSecondTranspose(permute); - other_shuffle->setName((layer_info(node) + "_IShuffleLayer_for_other").c_str()); - other = other_shuffle->getOutput(0); - LOG(INFO) << "after transpose other dim info turn to: " << other->getDimensions(); - - //self_T = self_dim == 2 ? self.t() : self.reshape({n, m}).t(); - if (self_dim == 1) { - //tensor1.reshape({n, m}) - std::vector new_shape; - new_shape = torch::reshape(torch::rand(origin_self_size), {n, m}).sizes().vec(); - auto tmp_shuffle = engine->network()->addShuffle(*self); - POROS_CHECK(tmp_shuffle, "Unable to create shuffle layer for MatmulConverter"); - tmp_shuffle->setReshapeDimensions(sizes_to_nvdim(new_shape)); - tmp_shuffle->setName((layer_info(node) + "_IShuffleLayer_for_self_tmp").c_str()); - self = tmp_shuffle->getOutput(0); - LOG(INFO) << "after reshape self dim info turn to: " << self->getDimensions(); - } - //self.t() - auto self_shuffle = engine->network()->addShuffle(*self); - POROS_CHECK(self_shuffle, "Unable to create shuffle layer for MatmulConverter"); - nvinfer1::Permutation first_perm; - first_perm.order[0] = 1; - first_perm.order[1] = 0; - self_shuffle->setFirstTranspose(first_perm); - self_shuffle->setZeroIsPlaceholder(false); - self_shuffle->setName((layer_info(node) + "_IShuffleLayer_for_self").c_str()); - self = self_shuffle->getOutput(0); - LOG(INFO) << "after transpose self dim info turn to: " << self->getDimensions(); - - //求 other.t() 与 self.t() 的matmul 的结果。 - auto mm_output = converter(engine, node, other, self); - POROS_CHECK(mm_output, "Unable to calculate transpose matmul for MatmulConverter"); - auto mm_dim = mm_output->getDimensions().nbDims; - auto mm_dim_size = nvdim_to_sizes(mm_output->getDimensions()); - - //给我转置回来... 要哭了 - if (self_dim == 2) { - std::vector new_order; - for (int i = 0; i < mm_dim; i++) { - new_order.push_back(i); - } - new_order[mm_dim - 1] = new_order[mm_dim - 2]; - new_order[mm_dim - 2] = mm_dim - 1; - auto mm_shuffle = engine->network()->addShuffle(*mm_output); - POROS_CHECK(mm_shuffle, "Unable to create shuffle layer for MatmulConverter"); - nvinfer1::Permutation permute; - std::copy(new_order.begin(), new_order.end(), permute.order); - mm_shuffle->setSecondTranspose(permute); - mm_shuffle->setName((layer_info(node) + "_IShuffleLayer_for_output").c_str()); - auto output = mm_shuffle->getOutput(0); - LOG(INFO) << "after transpose back ouput info turn to: " << output->getDimensions(); - return output; - } else { - //res_tensor.reshape(shape) - std::vector shape; - for (int i = 0; i < other_dim - 2; i++) { - shape.push_back(origin_other_size[i]); - } - shape.push_back(p); - auto new_shape = torch::reshape(torch::rand(mm_dim_size), shape).sizes().vec(); - auto mm_shuffle = engine->network()->addShuffle(*mm_output); - POROS_CHECK(mm_shuffle, "Unable to create shuffle layer for MatmulConverter"); - mm_shuffle->setReshapeDimensions(sizes_to_nvdim(new_shape)); - mm_shuffle->setName((layer_info(node) + "_IShuffleLayer_for_output").c_str()); - auto output = mm_shuffle->getOutput(0); - LOG(INFO) << "after transpose back ouput info turn to: " << output->getDimensions(); - return output; - } - - } else { - // expanding the dimensions if necessary. - if (self->getDimensions().nbDims < other->getDimensions().nbDims) { - auto newDims = self->getDimensions(); - for (int dim = self->getDimensions().nbDims; dim < other->getDimensions().nbDims; ++dim) { - newDims = unsqueeze_dims(newDims, 0, 1, false); - } - LOG(INFO) << "Original self shape: " << self->getDimensions() << ", reshaping to: " << newDims; - auto shuffle_layer = engine->network()->addShuffle(*self); - POROS_CHECK(shuffle_layer, "Unable to create shuffle layer for MatmulConverter"); - shuffle_layer->setReshapeDimensions(newDims); - shuffle_layer->setZeroIsPlaceholder(false); - shuffle_layer->setName((layer_info(node) + "_IShuffleLayer_for_self").c_str()); - self = shuffle_layer->getOutput(0); - //self = add_padding(engine, node, self, other->getDimensions().nbDims, false, false); - } else if (other->getDimensions().nbDims < self->getDimensions().nbDims) { - auto newDims = other->getDimensions(); - for (int dim = other->getDimensions().nbDims; dim < self->getDimensions().nbDims; ++dim) { - newDims = unsqueeze_dims(newDims, 0, 1, false); - } - LOG(INFO) << "Original other shape: " << other->getDimensions() << ", reshaping to: " << newDims; - auto shuffle_layer = engine->network()->addShuffle(*other); - POROS_CHECK(shuffle_layer, "Unable to create shuffle layer for MatmulConverter"); - shuffle_layer->setReshapeDimensions(newDims); - shuffle_layer->setZeroIsPlaceholder(false); - shuffle_layer->setName((layer_info(node) + "_IShuffleLayer_for_other").c_str()); - other = shuffle_layer->getOutput(0); - //other = add_padding(engine, node, other, self->getDimensions().nbDims, false, false); - } - - mm_layer = engine->network()->addMatrixMultiply( - *self, nvinfer1::MatrixOperation::kNONE, *other, nvinfer1::MatrixOperation::kNONE); - } - - mm_layer->setName((layer_info(node) + "_IMatrixMultiplyLayer_for_other").c_str()); - POROS_CHECK(mm_layer, "Unable to create matrix multiplication node: " << *node); - auto output = mm_layer->getOutput(0); - return output; -} - -bool MatmulConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for MatmulConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for MatmulConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[1]->type()->isSubtypeOf(c10::TensorType::get())), - "input[1] for MatmulConverter is not Tensor as expected"); - - auto self = engine->context().get_tensor(inputs[0]); - auto other = engine->context().get_tensor(inputs[1]); - POROS_CHECK_TRUE(((self != nullptr) && (other != nullptr)), - "Unable to init input tensor for node: " << *node); - - //add more log info for matmulConverter - LOG(INFO) << "input[0] tensor is: " << node_info(inputs[0]->node()); - LOG(INFO) << "input[1] tensor is: " << node_info(inputs[1]->node()); - - auto ouput = converter(engine, node, self, other); - if (ouput != nullptr) { - engine->context().set_tensor(node->outputs()[0], ouput); - LOG(INFO) << "Output tensor shape: " << ouput->getDimensions(); - return true; - } else { - return false; - } -} - -/* aten::bmm(Tensor self, Tensor mat2) -> Tensor */ -bool BmmConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for BmmConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for BmmConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[1]->type()->isSubtypeOf(c10::TensorType::get())), - "input[1] for BmmConverter is not Tensor as expected"); - - - auto self = engine->context().get_tensor(inputs[0]); - auto mat2 = engine->context().get_tensor(inputs[1]); - POROS_CHECK_TRUE(((self != nullptr) && (mat2 != nullptr)), - "Unable to init input tensor for node: " << *node); - - nvinfer1::Dims selfDims = self->getDimensions(); - nvinfer1::Dims mat2Dims = mat2->getDimensions(); - - // check dimensions - POROS_CHECK(selfDims.nbDims == 3, - "Expected 3-dimensional tensor, but got " << selfDims.nbDims - << "-dimensional tensor for argument #1 'batch1' (while checking arguments for bmm)"); - POROS_CHECK(mat2Dims.nbDims == 3, - "Expected 3-dimensional tensor, but got " << mat2Dims.nbDims - << "-dimensional tensor for argument #2 'batch2' (while checking arguments for bmm)"); - - // Self and mat2 should have same size at dimension 0 - POROS_CHECK(selfDims.d[0] == mat2Dims.d[0], - "Expected tensor to have size " << selfDims.d[0] << " at dimension 0, but got size " << mat2Dims.d[0] - << " for argument #2 'batch2' (while checking arguments for bmm)"); - - // The size of mat2 at dimension 1 should be the same as that of self at dimension 2. - POROS_CHECK(selfDims.d[2] == mat2Dims.d[1], - "Expected tensor to have size " << selfDims.d[2] << " at dimension 1, but got size " << mat2Dims.d[1] - << " for argument #2 'batch2' (while checking arguments for bmm)"); - - auto mm_layer = engine->network()->addMatrixMultiply( - *self, nvinfer1::MatrixOperation::kNONE, *mat2, nvinfer1::MatrixOperation::kNONE); - POROS_CHECK(mm_layer, "Unable to create matrix multiplication node: " << *node); - - mm_layer->setName((layer_info(node) + "_IMatrixMultiplyLayer").c_str()); - engine->context().set_tensor(node->outputs()[0], mm_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << mm_layer->getOutput(0)->getDimensions(); - return true; -} - -/** - * aten::addmm(Tensor self, Tensor mat1, Tensor mat2, *, Scalar beta=1, Scalar alpha=1) -> Tensor - * check the function in pytorch: aten/src/ATen/RegisterSparseCuda.cpp - * at::native::addmm_sparse_dense_cuda(self, mat1, mat2, beta, alpha) - * and the docs is like this: https://pytorch.org/docs/stable/generated/torch.addmm.html - * - * %out: Tensor = aten::addmm(%bias, %mat1, %mat2, %beta, %alpha) - * according to the torch.addmm explanation. the result is: - * out = %beta * %bias + %alpha (%mat1 @ %mat2 ) - * - * try to converter matmul like below: - * %mm: Tensor = aten::matmul(%mat1, %mat2) - * %bias_new: Tensor = aten::mul(%bias, %beta) - * %out: Tensor = aten::add(%bias_new, %mm, %alpha) - **/ -bool AddmmConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 5), "invaid inputs size for AddmmConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for AddmmConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[1]->type()->isSubtypeOf(c10::TensorType::get())), - "input[1] for AddmmConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[2]->type()->isSubtypeOf(c10::TensorType::get())), - "input[2] for AddmmConverter is not Tensor as expected"); - - //extract bias & mat1 & mat2 - auto bias = engine->context().get_tensor(inputs[0]); - auto mat1 = engine->context().get_tensor(inputs[1]); - auto mat2 = engine->context().get_tensor(inputs[2]); - POROS_CHECK_TRUE(((bias != nullptr) && (mat1 != nullptr) && (mat2 != nullptr)), - "Unable to init input tensor for node: " << *node); - - //extract beta & alpha - auto beta = (engine->context().get_constant(inputs[3])).toScalar().to(); - auto alpha = (engine->context().get_constant(inputs[4])).toScalar().to(); - - /*----------------------------------------------------------------------------- - step1: %mm: Tensor = aten::matmul(%mat1, %mat2) - -------------------------------------------------------------------------------*/ - // Ensure mat1 and mat2 tensors have same nbDims by expanding the dimensions (from 0 axis) if - // necessary. - // TODO: this is too much simpler than the reality. we should change this someday - if (mat1->getDimensions().nbDims < mat2->getDimensions().nbDims) { - mat1 = add_padding(engine, node, mat1, mat2->getDimensions().nbDims, false, false); - } else { - mat2 = add_padding(engine, node, mat2, mat1->getDimensions().nbDims, false, false); - } - auto mm_layer = engine->network()->addMatrixMultiply( - *mat1, nvinfer1::MatrixOperation::kNONE, *mat2, nvinfer1::MatrixOperation::kNONE); - POROS_CHECK(mm_layer, "Unable to create matrix multiplication node: " << *node); - mm_layer->setName((layer_info(node) + "_IMatrixMultiplyLayer").c_str()); - auto mm_output = mm_layer->getOutput(0); - - /*----------------------------------------------------------------------------- - step2: %bias_new: Tensor = aten::mul(%bias, %beta) - -------------------------------------------------------------------------------*/ - if (1 != beta) { - auto beta_tensor = tensor_to_const(engine, torch::tensor({beta})); - auto bias_new_layer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - bias, - beta_tensor, - layer_info(node) + "_prod_for_beta"); - POROS_CHECK(bias_new_layer, "Unable to create bias mul layer from node: " << *node); - bias = bias_new_layer->getOutput(0); - } - - /*----------------------------------------------------------------------------- - step3: %out: Tensor = aten::add(%bias_new, %mm, %alpha) - -------------------------------------------------------------------------------*/ - if (1 != alpha) { - auto alpha_tensor = tensor_to_const(engine, torch::tensor({alpha})); - auto mm_new_layer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - mm_output, - alpha_tensor, - layer_info(node) + "_prod_for_alpha"); - POROS_CHECK(mm_new_layer, "Unable to create alpha*input layer from node: " << *node); - mm_output = mm_new_layer->getOutput(0); - } - auto add_mm = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - bias, - mm_output, - layer_info(node) + "_sum"); - POROS_CHECK(add_mm, "Unable to create add layer from node: " << *node); - engine->context().set_tensor(node->outputs()[0], add_mm->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << add_mm->getOutput(0)->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, MatmulConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, BmmConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, AddmmConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/matrix_multiply.h b/poros/poros/converter/gpu/matrix_multiply.h deleted file mode 100644 index 4327cc89bb6..00000000000 --- a/poros/poros/converter/gpu/matrix_multiply.h +++ /dev/null @@ -1,104 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file matrix_multiply.h -* @author tianjinjin@baidu.com -* @date Wed Aug 18 20:30:19 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class MatmulConverter : public GpuConverter { -public: - MatmulConverter() {} - virtual ~MatmulConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - nvinfer1::ITensor* converter(TensorrtEngine* engine, - const torch::jit::Node *node, - nvinfer1::ITensor* self, - nvinfer1::ITensor* other); - - const std::vector schema_string() { - return {"aten::matmul(Tensor self, Tensor other) -> Tensor"}; - } - - /** - * TODO: TRY TO SUPPORT SCHEMA PATTERNS BELLOW: - * aten::matmul.out(Tensor self, Tensor other, *, Tensor(a!) out) -> Tensor(a!) - * **/ - const std::vector node_kind() { - return {torch::jit::aten::matmul}; - } -}; - -class BmmConverter : public GpuConverter { -public: - BmmConverter() {} - virtual ~BmmConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::bmm(Tensor self, Tensor mat2) -> Tensor"}; - } - - /** - * TODO: TRY TO SUPPORT SCHEMA PATTERNS BELLOW: - * aten::bmm.out(Tensor self, Tensor mat2, *, Tensor(a!) out) -> Tensor(a!) - * **/ - const std::vector node_kind() { - return {torch::jit::aten::bmm}; - } -}; - -class AddmmConverter : public GpuConverter { -public: - AddmmConverter() {} - virtual ~AddmmConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::addmm(Tensor self, Tensor mat1, Tensor mat2, *, Scalar beta=1, Scalar alpha=1) -> Tensor"}; - } - - /** - * TODO: TRY TO SUPPORT SCHEMA PATTERNS BELLOW: - * aten::addmm.out(Tensor self, Tensor mat1, Tensor mat2, *, Scalar beta=1, Scalar alpha=1, Tensor(a!) out) -> Tensor(a!) - * aten::addmm_(Tensor(a!) self, Tensor mat1, Tensor mat2, *, Scalar beta=1, Scalar alpha=1) -> Tensor(a!) - * **/ - const std::vector node_kind() { - return {torch::jit::aten::addmm}; - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/converter/gpu/meshgrid.cpp b/poros/poros/converter/gpu/meshgrid.cpp deleted file mode 100644 index b8a1e3b2962..00000000000 --- a/poros/poros/converter/gpu/meshgrid.cpp +++ /dev/null @@ -1,140 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file meshgrid.cpp -* @author wangrui39@baidu.com -* @date Monday November 27 11:36:11 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/meshgrid.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -// aten::meshgrid(Tensor[] tensors) -> Tensor[] -bool MeshgridConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 1), "invaid inputs size for MeshgridConverter"); - - // 1:获取输入 - std::vector tensorlist; - //tensorlist.resize(2); - POROS_CHECK_TRUE((engine->context().get_tensorlist(inputs[0], tensorlist)), "extract tensor list err"); - - POROS_CHECK_TRUE((tensorlist.size() == 2), - "Expected 2 elements in a tensorlist but found " + std::to_string(tensorlist.size())); - nvinfer1::ITensor* input1 = tensorlist[0]; - POROS_CHECK_TRUE((input1->getDimensions().nbDims == 1), - "Expected scalar or 1D tensor in the tensor list but got " + std::to_string(input1->getDimensions().nbDims)); - nvinfer1::ITensor* input2 = tensorlist[1]; - POROS_CHECK_TRUE((input1->getDimensions().nbDims == 1), - "Expected scalar or 1D tensor in the tensor list but got " + std::to_string(input2->getDimensions().nbDims)); - - /*std::vector output_tensorlist; - output_tensorlist.emplace_back(input1); - output_tensorlist.emplace_back(input2);*/ - - // 2: 构造返回类型 - nvinfer1::Dims reshape_dim; - reshape_dim.nbDims = 2; - reshape_dim.d[0] = 1; - reshape_dim.d[1] = input1->getDimensions().d[0]; - std::vector output_tensorlist; - output_tensorlist.resize(2); - - // 3:生成return tensorlist[0] unsqueeze + cat + transpose - // a:unsqueeze - auto unsqueeze_shuffle_layer1 = engine->network()->addShuffle(*input1); - POROS_CHECK(unsqueeze_shuffle_layer1, "Unable to create shuffle layer from node: " << *node); - unsqueeze_shuffle_layer1->setReshapeDimensions(reshape_dim); - unsqueeze_shuffle_layer1->setName((layer_info(node) + "_IShuffleLayer_for_input1").c_str()); - nvinfer1::ITensor *un_sl_output1 = unsqueeze_shuffle_layer1->getOutput(0); - - // b:cat - std::vector cat_tensorlist1; - cat_tensorlist1.resize(input2->getDimensions().d[0]); - for (int i = 0; i < input2->getDimensions().d[0]; ++i) { - auto tmp_weights = Weights(at::zeros({un_sl_output1->getDimensions().d[0], un_sl_output1->getDimensions().d[1]}, {at::kCUDA}).to(torch::kInt)); - auto constant_layer = engine->network()->addConstant(tmp_weights.shape, tmp_weights.data); - nvinfer1::ITensor* costant_tensor = constant_layer->getOutput(0); - auto add_layer = engine->network()->addElementWise(*costant_tensor, *un_sl_output1, nvinfer1::ElementWiseOperation::kSUM); - add_layer->setName((layer_info(node) + "_sum_for_tensorlist1_" + std::to_string(i)).c_str()); - cat_tensorlist1[i] = add_layer->getOutput(0); - } - - auto cat_layer1 = engine->network()->addConcatenation(cat_tensorlist1.data(), cat_tensorlist1.size()); - cat_layer1->setAxis(0); - cat_layer1->setName((layer_info(node) + "_IConcatenationLayer_1").c_str()); - nvinfer1::ITensor *cat_output1 = cat_layer1->getOutput(0); - - // c:transpose - auto transpose_shuffle_layer = engine->network()->addShuffle(*cat_output1); - POROS_CHECK(transpose_shuffle_layer, "Unable to create shuffle layer from node: " << *node); - nvinfer1::Permutation permute; - permute.order[0] = 1; - permute.order[1] = 0; - transpose_shuffle_layer->setSecondTranspose(permute); - transpose_shuffle_layer->setName((layer_info(node) + "_IShuffleLayer_for_cat_output").c_str()); - nvinfer1::ITensor *ts_output = transpose_shuffle_layer->getOutput(0); - output_tensorlist[0] = ts_output; - - // 4:生成return tensorlist[1] unsqueeze + cat - // a:unsqueeze - reshape_dim.d[1] = input2->getDimensions().d[0]; - auto unsqueeze_shuffle_layer2 = engine->network()->addShuffle(*input2); - POROS_CHECK(unsqueeze_shuffle_layer2, "Unable to create shuffle layer from node: " << *node); - unsqueeze_shuffle_layer2->setReshapeDimensions(reshape_dim); - unsqueeze_shuffle_layer2->setName((layer_info(node) + "_IShuffleLayer_for_input2").c_str()); - nvinfer1::ITensor *un_sl_output2 = unsqueeze_shuffle_layer2->getOutput(0); - - // b:cat - std::vector cat_tensorlist2; - cat_tensorlist2.resize(input1->getDimensions().d[0]); - for (int i = 0; i < input1->getDimensions().d[0]; ++i) { - auto tmp_weights = Weights(at::zeros({un_sl_output2->getDimensions().d[0], un_sl_output2->getDimensions().d[1]}, {at::kCUDA}).to(torch::kInt)); - auto constant_layer = engine->network()->addConstant(tmp_weights.shape, tmp_weights.data); - nvinfer1::ITensor* costant_tensor = constant_layer->getOutput(0); - auto add_layer = engine->network()->addElementWise(*costant_tensor, *un_sl_output2, nvinfer1::ElementWiseOperation::kSUM); - //cat_tensorlist2.emplace_back(add_layer->getOutput(0)); - add_layer->setName((layer_info(node) + "_sum_for_tensorlist2_" + std::to_string(i)).c_str()); - cat_tensorlist2[i] = add_layer->getOutput(0); - } - - auto cat_layer2 = engine->network()->addConcatenation(cat_tensorlist2.data(), cat_tensorlist2.size()); - cat_layer2->setAxis(0); - cat_layer2->setName((layer_info(node) + "_IConcatenationLayer_2").c_str()); - nvinfer1::ITensor *cat_output2 = cat_layer2->getOutput(0); - output_tensorlist[1] = cat_output2; - - // 5:设置output - engine->context().set_tensorlist(node->outputs()[0], output_tensorlist); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, MeshgridConverter); - -} // baidu -} // mirana -} // poros diff --git a/poros/poros/converter/gpu/meshgrid.h b/poros/poros/converter/gpu/meshgrid.h deleted file mode 100644 index b29d6c4e5e3..00000000000 --- a/poros/poros/converter/gpu/meshgrid.h +++ /dev/null @@ -1,61 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file meshgrid.h -* @author wangrui39@baidu.com -* @date Monday November 27 11:36:11 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -// DEPRECATED Use `lowering/fuse_meshgrid.h` to rewrite this op. This converter is no longer needed. -// Correspons to torch.meshgrid https://pytorch.org/docs/1.9.0/generated/torch.meshgrid.html?highlight=meshgrid#torch.meshgrid -class MeshgridConverter : public GpuConverter { -public: - MeshgridConverter() {} - virtual ~MeshgridConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::meshgrid(Tensor[] tensors) -> Tensor[]"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::meshgrid}; - } - - bool assign_schema_attr() { - return assign_schema_attr_helper({{"aten::meshgrid(Tensor[] tensors) -> Tensor[]", {0, 0}}}); - } -}; - - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/mul_div.cpp b/poros/poros/converter/gpu/mul_div.cpp deleted file mode 100644 index 51aa0ca224d..00000000000 --- a/poros/poros/converter/gpu/mul_div.cpp +++ /dev/null @@ -1,297 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file mul_div.cpp -* @author tianjinjin@baidu.com -* @date Mon Mar 8 11:36:11 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/mul_div.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/* -aten::mul.Tensor(Tensor self, Tensor other) -> Tensor -aten::mul.Scalar(Tensor self, Scalar other) -> Tensor*/ -bool MulConverter::converter(TensorrtEngine *engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for MulConverter"); - - // 先判断schema是否是 aten::mul.int(int a, int b) -> (int) - if (node->schema().operator_name() == torch::jit::parseSchema(this->schema_string()[4]).operator_name()) { - // 检查int是否以nvtensor的形式输入 - if (check_inputs_tensor_scalar(engine, node)) { - // 获取int对应的nvtensor - nvinfer1::ITensor *a = this->get_tensor_scalar(inputs[0]); - nvinfer1::ITensor *b = this->get_tensor_scalar(inputs[1]); - // 判断是否为空 (get_constant失败时可能为空) - // 为空时返回false, 让子图fallback - POROS_CHECK_TRUE((a != nullptr && b != nullptr), - node_info(node) + std::string("get int nvtensor false.")); - // a和b相乘并返回 - nvinfer1::ILayer *mul_layer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - a, b, layer_info(node) + "_prod"); - POROS_CHECK(mul_layer, "Unable to create mul layer from node: " << *node); - nvinfer1::ITensor *output = mul_layer->getOutput(0); - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - return true; - } else { - int a = engine->context().get_constant(inputs[0]).toScalar().to(); - int b = engine->context().get_constant(inputs[1]).toScalar().to(); - engine->context().set_constant(node->outputs()[0], a * b); - return true; - } - } - - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for MulConverter is not Tensor as expected"); - - // Should implement self * other - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - auto other = engine->context().get_tensor(inputs[1]); - // when other input is Scalar - if (other == nullptr) { - auto other_const = engine->context().get_constant(inputs[1]); - if (other_const.isScalar()) { - auto other_scalar = other_const.toScalar().to(); - other = tensor_to_const(engine, torch::tensor({other_scalar})); - } else { - POROS_THROW_ERROR("Unable to get input other value for MulConverter"); - } - } - // 遇到过aten::mul(float tensor, int scalar)的输入情况,都转成float就行 - if (self->getType() == nvinfer1::DataType::kFLOAT && other->getType() == nvinfer1::DataType::kINT32) { - nvinfer1::IIdentityLayer* identity_layer = engine->network()->addIdentity(*other); - identity_layer->setOutputType(0, nvinfer1::DataType::kFLOAT); - identity_layer->setName((layer_info(node) + "_IIdentityLayer_for_other").c_str()); - other = identity_layer->getOutput(0); - } else if (other->getType() == nvinfer1::DataType::kFLOAT && self->getType() == nvinfer1::DataType::kINT32) { - nvinfer1::IIdentityLayer* identity_layer = engine->network()->addIdentity(*self); - identity_layer->setOutputType(0, nvinfer1::DataType::kFLOAT); - identity_layer->setName((layer_info(node) + "_IIdentityLayer_for_self").c_str()); - self = identity_layer->getOutput(0); - } - - auto mul = add_elementwise(engine, nvinfer1::ElementWiseOperation::kPROD, self, other, layer_info(node) + "_prod"); - POROS_CHECK(mul, "Unable to create mul layer from node: " << *node); - engine->context().set_tensor(node->outputs()[0], mul->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << mul->getOutput(0)->getDimensions(); - return true; -} - -/* -aten::div.Tensor(Tensor self, Tensor other) -> Tensor -aten::div.Scalar(Tensor self, Scalar other) -> Tensor*/ -bool DivConverter::converter(TensorrtEngine *engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for DivConverter"); - - // aten::div.int(int a, int b) -> (float) - if (node->schema().operator_name() == torch::jit::parseSchema(this->schema_string()[4]).operator_name() || \ - node->schema().operator_name() == torch::jit::parseSchema(this->schema_string()[5]).operator_name()) { - if (check_inputs_tensor_scalar(engine, node)) { - nvinfer1::ITensor *a = this->get_tensor_scalar(inputs[0]); - nvinfer1::ITensor *b = this->get_tensor_scalar(inputs[1]); - POROS_CHECK_TRUE((a != nullptr && b != nullptr), - node_info(node) + std::string("get int nvtensor false.")); - // Set datatype for input tensor to kFLOAT - auto identity_layer1 = engine->network()->addIdentity(*a); - identity_layer1->setOutputType(0, nvinfer1::DataType::kFLOAT); - identity_layer1->setName((layer_info(node) + "_IIdentityLayer_for_input0").c_str()); - nvinfer1::ITensor *a_float = identity_layer1->getOutput(0); - // Set datatype for input tensor to kFLOAT - auto identity_layer2 = engine->network()->addIdentity(*b); - identity_layer2->setOutputType(0, nvinfer1::DataType::kFLOAT); - identity_layer2->setName((layer_info(node) + "_IIdentityLayer_for_input1").c_str()); - nvinfer1::ITensor *b_float = identity_layer2->getOutput(0); - - nvinfer1::ILayer *div_layer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kDIV, - a_float, b_float, layer_info(node) + "_div"); - POROS_CHECK(div_layer, "Unable to create div layer from node: " << *node); - nvinfer1::ITensor *output = div_layer->getOutput(0); - LOG(INFO) << "div output type: " << output->getType(); - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - return true; - } else { - int a = engine->context().get_constant(inputs[0]).toScalar().to(); - int b = engine->context().get_constant(inputs[1]).toScalar().to(); - float output = float(a) / float(b); - engine->context().set_constant(node->outputs()[0], output); - return true; - } - } - - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for DivConverter is not Tensor as expected"); - - // Should implement self / other - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - auto other = engine->context().get_tensor(inputs[1]); - //when other input is Scalar - if (other == nullptr) { - auto other_const = engine->context().get_constant(inputs[1]); - if (other_const.isScalar()) { - auto other_scalar = other_const.toScalar().to(); - other = tensor_to_const(engine, torch::tensor({other_scalar})); - } else { - POROS_THROW_ERROR("Unable to get input other value for DivConverter"); - } - } - - if (self->getType() == nvinfer1::DataType::kINT32) { - nvinfer1::IIdentityLayer* identity_self_layer = engine->network()->addIdentity(*self); - identity_self_layer->setOutputType(0, nvinfer1::DataType::kFLOAT); - self = identity_self_layer->getOutput(0); - } - - if (other->getType() == nvinfer1::DataType::kINT32) { - nvinfer1::IIdentityLayer* identity_other_layer = engine->network()->addIdentity(*other); - identity_other_layer->setOutputType(0, nvinfer1::DataType::kFLOAT); - other = identity_other_layer->getOutput(0); - } - - auto div = add_elementwise(engine, nvinfer1::ElementWiseOperation::kDIV, self, other, layer_info(node) + "_div"); - POROS_CHECK(div, "Unable to create div layer from node: " << *node); - engine->context().set_tensor(node->outputs()[0], div->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << div->getOutput(0)->getDimensions(); - return true; -} - -// aten::floordiv.int(int a, int b) -> (int) -// aten::__round_to_zero_floordiv(int a, int b) -> (int) -bool FloordivConverter::converter(TensorrtEngine *engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for ScalarFloordivConverter"); - - // 输入是nvtensor - if (check_inputs_tensor_scalar(engine, node)) { - // __round_to_zero_div 待支持 - nvinfer1::ITensor *a = this->get_tensor_scalar(inputs[0]); - nvinfer1::ITensor *b = this->get_tensor_scalar(inputs[1]); - - POROS_CHECK_TRUE((a != nullptr && b != nullptr), - node_info(node) + std::string("get int nvtensor false.")); - - nvinfer1::ElementWiseOperation opreation; - std::string nv_layer_name; - if (node->schema().operator_name() == torch::jit::parseSchema(this->schema_string()[1]).operator_name()) { - opreation = nvinfer1::ElementWiseOperation::kDIV; - nv_layer_name = layer_info(node) + "_div"; - } else { - opreation = nvinfer1::ElementWiseOperation::kFLOOR_DIV; - nv_layer_name = layer_info(node) + "_floor_div"; - } - - nvinfer1::ILayer *floordiv_layer = add_elementwise(engine, - opreation, - a, b, nv_layer_name); - POROS_CHECK(floordiv_layer, "Unable to create floordiv layer from node: " << *node); - nvinfer1::ITensor *output = floordiv_layer->getOutput(0); - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - } else { - // 输入是int ivalue - int a = engine->context().get_constant(inputs[0]).toScalar().to(); - int b = engine->context().get_constant(inputs[1]).toScalar().to(); - POROS_CHECK_TRUE((b != 0), "invaid inputs[1] for ScalarFloordivConverter, which is equal to zero"); - int output = 0; - if (node->kind() == node_kind()[0]) { - output = std::floor(float(a) / float(b)); - } else { - output = int(a / b); - } - engine->context().set_constant(node->outputs()[0], output); - } - return true; -} - -//aten::remainder.Scalar(Tensor self, Scalar other) -> (Tensor) -//aten::remainder.Tensor(Tensor self, Tensor other) -> (Tensor) -bool RemainderConverter::converter(TensorrtEngine *engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for RemainderConverter"); - - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for RemainderConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[1]->type()->kind() == c10::TypeKind::FloatType || - inputs[1]->type()->kind() == c10::TypeKind::IntType || - inputs[1]->type()->kind() == c10::TypeKind::TensorType), - "input[1] for RemainderConverter is not Scalar as expected"); - - nvinfer1::ITensor *self = engine->context().get_tensor(inputs[0]); - - nvinfer1::ITensor *other; - - if (inputs[1]->type()->kind() == c10::TypeKind::TensorType) { - other = engine->context().get_tensor(inputs[1]); - } else { - other = tensor_to_const(engine, - torch::tensor(engine->context().get_constant(inputs[1]).toDouble(), torch::kFloat32)); - } - - POROS_CHECK_TRUE((self != nullptr && other != nullptr), - node_info(node) + std::string("get int nvtensor false.")); - - // floor_div - nvinfer1::ILayer *floordiv_layer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kFLOOR_DIV, - self, other, layer_info(node) + "_floor_div"); - POROS_CHECK(floordiv_layer, "Unable to create floordiv layer from node: " << *node); - nvinfer1::ITensor *floordiv_output = floordiv_layer->getOutput(0); - - // prod - nvinfer1::ILayer *prod_layer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - floordiv_output, other, layer_info(node) + "_prod"); - POROS_CHECK(prod_layer, "Unable to create prod layer from node: " << *node); - nvinfer1::ITensor *prod_output = prod_layer->getOutput(0); - - // sub - nvinfer1::ILayer *sub_layer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUB, - self, prod_output, layer_info(node) + "_sub"); - POROS_CHECK(sub_layer, "Unable to create sub layer from node: " << *node); - nvinfer1::ITensor *output = sub_layer->getOutput(0); - - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, MulConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, DivConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, FloordivConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, RemainderConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/mul_div.h b/poros/poros/converter/gpu/mul_div.h deleted file mode 100644 index 20da0a20e7c..00000000000 --- a/poros/poros/converter/gpu/mul_div.h +++ /dev/null @@ -1,160 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file mul_div.h -* @author tianjinjin@baidu.com -* @date Mon Aug 16 12:26:28 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class MulConverter : public GpuConverter { -public: - MulConverter() {} - virtual ~MulConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::mul.Tensor(Tensor self, Tensor other) -> Tensor", - "aten::mul.Scalar(Tensor self, Scalar other) -> Tensor", - "aten::mul_.Tensor(Tensor(a!) self, Tensor other) -> Tensor(a!)", - "aten::mul_.Scalar(Tensor(a!) self, Scalar other) -> Tensor(a!)", - "aten::mul.int(int a, int b) -> (int)", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::mul.out(Tensor self, Tensor other, *, Tensor(a!) out) -> Tensor(a!)", - * **/ - const std::vector node_kind() { - return {torch::jit::aten::mul, - torch::jit::aten::mul_}; - } - - bool assign_schema_attr() { - bool result = true; - result &= assign_schema_attr_helper({{"aten::mul.int(int a, int b) -> (int)", {1, 1}}}); - result &= assign_schema_attr_helper({{"aten::mul.Scalar(Tensor self, Scalar other) -> Tensor", {1, 1}}}); - result &= assign_schema_attr_helper({{"aten::mul_.Scalar(Tensor(a!) self, Scalar other) -> Tensor(a!)", {1, 1}}}); - return result; - } -}; - -class DivConverter : public GpuConverter { -public: - DivConverter() {} - virtual ~DivConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::div.Tensor(Tensor self, Tensor other) -> Tensor", - "aten::div.Scalar(Tensor self, Scalar other) -> (Tensor)", - "aten::div_.Tensor(Tensor(a!) self, Tensor other) -> Tensor(a!)", - "aten::div_.Scalar(Tensor(a!) self, Scalar other) -> Tensor(a!)", - "aten::div.int(int a, int b) -> (float)", - "aten::div(Scalar a, Scalar b) -> (float)" - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::div.out(Tensor self, Tensor other, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::div.Tensor_mode(Tensor self, Tensor other, *, str? rounding_mode) -> Tensor", - * "aten::div.Scalar_mode(Tensor self, Scalar other, *, str? rounding_mode) -> Tensor" - * "aten::div.out_mode(Tensor self, Tensor other, *, str? rounding_mode, Tensor(a!) out) -> Tensor(a!)" - * **/ - const std::vector node_kind() { - return {torch::jit::aten::div, - torch::jit::aten::div_}; - } - - bool assign_schema_attr() { - bool result = true; - result &= assign_schema_attr_helper({{"aten::div.int(int a, int b) -> (float)", {1, 1}}}); - result &= assign_schema_attr_helper({{"aten::div(Scalar a, Scalar b) -> (float)", {1, 1}}}); - return result; - } - -}; - -class FloordivConverter : public GpuConverter { -public: - FloordivConverter() {} - virtual ~FloordivConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::floordiv.int(int a, int b) -> (int)", - "aten::__round_to_zero_floordiv.int(int a, int b) -> (int)" - }; - } - - const std::vector node_kind() { - return {torch::jit::aten::floordiv, - torch::jit::aten::__round_to_zero_floordiv}; - } - - bool assign_schema_attr() { - bool result = true; - result &= assign_schema_attr_helper({{"aten::floordiv.int(int a, int b) -> (int)", {1, 1}}}); - result &= assign_schema_attr_helper({{"aten::__round_to_zero_floordiv.int(int a, int b) -> (int)", {1, 1}}}); - return result; - } -}; - - -class RemainderConverter : public GpuConverter { -public: - RemainderConverter() {} - virtual ~RemainderConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::remainder.Scalar(Tensor self, Scalar other) -> (Tensor)", - "aten::remainder.Tensor(Tensor self, Tensor other) -> (Tensor)", - }; - } - - const std::vector node_kind() { - return {torch::jit::aten::remainder}; - } - - bool assign_schema_attr() { - bool result = true; - result &= assign_schema_attr_helper({{"aten::remainder.Scalar(Tensor self, Scalar other) -> (Tensor)", {1, 1}}}); - result &= assign_schema_attr_helper({{"aten::remainder.Tensor(Tensor self, Tensor other) -> (Tensor)", {1, 1}}}); - return result; - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/non_converterable.cpp b/poros/poros/converter/gpu/non_converterable.cpp deleted file mode 100644 index 1212799d4c4..00000000000 --- a/poros/poros/converter/gpu/non_converterable.cpp +++ /dev/null @@ -1,118 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file non_converterable.cpp -* @author tianjinjin@baidu.com -* @date Thu Aug 26 10:24:14 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/non_converterable.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/converter/gpu/weight.h" -#include "poros/context/poros_global.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/*aten::contiguous(Tensor(a) self, *, MemoryFormat memory_format=contiguous_format) -> Tensor(a)*/ -bool ContiguousConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE(inputs[0]->type()->isSubtypeOf(c10::TensorType::get()), - "input[0] for ContiguousConverter is not Tensor as expected"); - - //extract tensors - auto in_tensor = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in_tensor != nullptr), "Unable to init input tensor for node: " << *node); - //need to do nothing, update the map directly. - engine->context().set_tensor(node->outputs()[0], in_tensor); - LOG(INFO) << "Output tensor shape: " << in_tensor->getDimensions(); - return true; -} - -/*aten::dropout(Tensor input, float p, bool train) -> Tensor*/ -bool DropoutConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 3), "invaid inputs size for DropoutConverter"); - POROS_CHECK_TRUE(inputs[0]->type()->isSubtypeOf(c10::TensorType::get()), - "input[0] for DropoutConverter is not Tensor as expected"); - - //extract tensors - auto in_tensor = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in_tensor != nullptr), "Unable to init input tensor for node: " << *node); - //need to do nothing, update the map directly. - engine->context().set_tensor(node->outputs()[0], in_tensor); - LOG(INFO) << "Output tensor shape: " << in_tensor->getDimensions(); - return true; -} - -// aten::IntImplicit(Tensor a) -> (int) -bool IntimplicitConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE(inputs[0]->type()->isSubtypeOf(c10::TensorType::get()), - "input[0] for ContiguousConverter is not Tensor as expected"); - - //extract tensors - auto in_tensor = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in_tensor != nullptr), "Unable to init input tensor for node: " << *node); - //need to do nothing, update the map directly. - engine->context().set_tensor(node->outputs()[0], in_tensor); - LOG(INFO) << "Output tensor shape: " << in_tensor->getDimensions(); - return true; -} - -// prim::tolist(Tensor a) -> (int[]) -bool TolistConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE(inputs[0]->type()->isSubtypeOf(c10::TensorType::get()), - "input[0] for ContiguousConverter is not Tensor as expected"); - - //extract tensors - auto in_tensor = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in_tensor != nullptr), "Unable to init input tensor for node: " << *node); - //need to do nothing, update the map directly. - engine->context().set_tensor(node->outputs()[0], in_tensor); - LOG(INFO) << "Output tensor shape: " << in_tensor->getDimensions(); - return true; -} - -// aten::detach(Tensor(a) self) -> Tensor(a) -bool DetachConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE(inputs[0]->type()->isSubtypeOf(c10::TensorType::get()), - "input[0] for DetachConverter is not Tensor as expected"); - - //extract tensors - auto in_tensor = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in_tensor != nullptr), "Unable to init input tensor for node: " << *node); - //need to do nothing, update the map directly. - engine->context().set_tensor(node->outputs()[0], in_tensor); - LOG(INFO) << "Output tensor shape: " << in_tensor->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, ContiguousConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, DropoutConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, IntimplicitConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, TolistConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, DetachConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/non_converterable.h b/poros/poros/converter/gpu/non_converterable.h deleted file mode 100644 index 7c82fd67004..00000000000 --- a/poros/poros/converter/gpu/non_converterable.h +++ /dev/null @@ -1,141 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file non_converterable.h -* @author tianjinjin@baidu.com -* @date Thu Aug 26 10:24:14 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class ContiguousConverter : public GpuConverter { -public: - ContiguousConverter() {} - virtual ~ContiguousConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::contiguous(Tensor(a) self, *, MemoryFormat memory_format=contiguous_format) -> Tensor(a)"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::contiguous}; - } -}; - -class DropoutConverter : public GpuConverter { -public: - DropoutConverter() {} - virtual ~DropoutConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::dropout(Tensor input, float p, bool train) -> Tensor", - "aten::dropout_(Tensor(a!) self, float p, bool train) -> Tensor(a!)", - "aten::feature_dropout(Tensor input, float p, bool train) -> Tensor", - "aten::feature_alpha_dropout(Tensor input, float p, bool train) -> Tensor", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::feature_dropout_(Tensor(a!) self, float p, bool train) -> Tensor(a!)", - * "aten::feature_alpha_dropout_(Tensor(a!) self, float p, bool train) -> Tensor(a!)", - * - * some stange err msg like : feature_alpha_dropout_ is not a member of 'torch::jit::aten' - * - * **/ - const std::vector node_kind() { - return {torch::jit::aten::dropout, - torch::jit::aten::dropout_, - torch::jit::aten::feature_dropout, - torch::jit::aten::feature_alpha_dropout,}; - } - -}; - -// aten::IntImplicit(Tensor a) -> (int) -class IntimplicitConverter : public GpuConverter { -public: - IntimplicitConverter() {} - virtual ~IntimplicitConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::IntImplicit(Tensor a) -> (int)"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::IntImplicit}; - } - -}; - -// prim::tolist -class TolistConverter : public GpuConverter { -public: - TolistConverter() {} - virtual ~TolistConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - //prim::tolist kind node has no schema - const std::vector schema_string() { - return {}; - } - - const std::vector node_kind() { - return {torch::jit::prim::tolist}; - } - -}; - -// aten::detach -class DetachConverter : public GpuConverter { -public: - DetachConverter() {} - virtual ~DetachConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - //prim::tolist kind node has no schema - const std::vector schema_string() { - return {"aten::detach(Tensor(a) self) -> Tensor(a)"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::detach}; - } - -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/norm.cpp b/poros/poros/converter/gpu/norm.cpp deleted file mode 100644 index 5d47b1a1aff..00000000000 --- a/poros/poros/converter/gpu/norm.cpp +++ /dev/null @@ -1,183 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file norm.cpp -* @author Lin Xiao Chun (linxiaochun@baidu.com) -* @date 2022-02-23 20:33:41 -* @brief -**/ - -#include -#include "poros/converter/gpu/norm.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -bool NormConverter::converter(TensorrtEngine *engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - // inputs.size() == 4 - POROS_CHECK_TRUE((inputs.size() == 4), "invaid inputs size for NormConverter") - - // inputs[0] => self - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for NormConverter is not Tensor as expected"); - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node) - auto self_dims = self->getDimensions(); - - // inputs[1] => p - POROS_CHECK_TRUE((inputs[1]->node()->kind() == torch::jit::prim::Constant), - "Non-constant p is not support for NormConverter"); - auto p_const = engine->context().get_constant(inputs[1]); - POROS_CHECK_TRUE((p_const.isScalar()), "Non-scalar p is not support for NormConverter") - nvinfer1::ITensor *p, *p_inverse; - auto p_scalar = p_const.toScalar().to(); - p = tensor_to_const(engine, torch::tensor(p_scalar)); - p_inverse = tensor_to_const(engine, torch::tensor(1.0 / p_scalar)); - - // inputs[2] => dims - POROS_CHECK_TRUE((inputs[2]->node()->kind() == torch::jit::prim::Constant), - "Non-constant dims is not support for NormConverter"); - auto dims_const = engine->context().get_constant(inputs[2]); - POROS_CHECK_TRUE((dims_const.isIntList()), " dims type must be int[] for NormConverter") - auto dims_list = dims_const.toIntList().vec(); - uint32_t dim = 0; - for (auto d: dims_list) { - if (d < 0) { - d = self_dims.nbDims + d; - } - dim |= 1 << d; - } - if (dim == 0) { - dim = pow(2, self_dims.nbDims) - 1; - } - - // input[3] => keepdim - POROS_CHECK_TRUE((inputs[3]->node()->kind() == torch::jit::prim::Constant), - "Non-constant dims is not support for NormConverter"); - auto keepdim_const = engine->context().get_constant(inputs[3]); - POROS_CHECK_TRUE((keepdim_const.isBool()), " dims type must be int[] for NormConverter") - auto keepdim = keepdim_const.toBool(); - - // unary_layer - auto unary_layer = engine->network()->addUnary(*self, nvinfer1::UnaryOperation ::kABS); - unary_layer->setName((layer_info(node) + "_IUnaryLayer").c_str()); - auto unary_output = unary_layer->getOutput(0); - - // elementwise_layer 1 - auto ew1_layer = add_elementwise(engine, nvinfer1::ElementWiseOperation::kPOW, unary_output, p, - layer_info(node) + "_pow_for_unary"); - - POROS_CHECK(ew1_layer, "Unable to create POW layer from node: " << *node); - auto ew_output = ew1_layer->getOutput(0); - - // reduce_layer - auto reduce_layer = engine->network()->addReduce(*ew_output, nvinfer1::ReduceOperation::kSUM, dim, keepdim); - reduce_layer->setName((layer_info(node) + "_IReduceLayer").c_str()); - auto reduce_output = reduce_layer->getOutput(0); - - // elementwise_layer 2 - auto ew2_layer = add_elementwise(engine, nvinfer1::ElementWiseOperation::kPOW, reduce_output, p_inverse, - layer_info(node) + "_pow_for_reduce"); - - engine->context().set_tensor(node->outputs()[0], ew2_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << ew2_layer->getOutput(0)->getDimensions(); - return true; -} - -bool FrobeniusNormConverter::converter(TensorrtEngine *engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - // inputs.size() == 3 - POROS_CHECK_TRUE((inputs.size() == 3), "invaid inputs size for FrobeniusNormConverter") - - // inputs[0] => self - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for FrobeniusNormConverter is not Tensor as expected"); - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node) - auto self_dims = self->getDimensions(); - - // p - nvinfer1::ITensor *p, *p_inverse; - float p_scalar = 2; - p = tensor_to_const(engine, torch::tensor(p_scalar)); - p_inverse = tensor_to_const(engine, torch::tensor(1.0 / p_scalar)); - - // inputs[1] => dims - POROS_CHECK_TRUE((inputs[1]->node()->kind() == torch::jit::prim::Constant), - "Non-constant dims is not support for FrobeniusNormConverter"); - auto dims_const = engine->context().get_constant(inputs[1]); - POROS_CHECK_TRUE((dims_const.isIntList()), " dims type must be int[] for FrobeniusNormConverter") - auto dims_list = dims_const.toIntList().vec(); - uint32_t dim = 0; - for (auto d: dims_list) { - if (d < 0) { - d = self_dims.nbDims + d; - } - dim |= 1 << d; - } - // in case of dims_list is empty or invalid, reduce on all axes - if (dim == 0) { - dim = pow(2, self_dims.nbDims) - 1; - } - - // input[2] => keepdim - POROS_CHECK_TRUE((inputs[2]->node()->kind() == torch::jit::prim::Constant), - "Non-constant dims is not support for FrobeniusNormConverter"); - auto keepdim_const = engine->context().get_constant(inputs[2]); - POROS_CHECK_TRUE((keepdim_const.isBool()), " dims type must be int[] for FrobeniusNormConverter") - auto keepdim = keepdim_const.toBool(); - - // unary_layer - auto unary_layer = engine->network()->addUnary(*self, nvinfer1::UnaryOperation ::kABS); - unary_layer->setName((layer_info(node) + "_IUnaryLayer").c_str()); - auto unary_output = unary_layer->getOutput(0); - - // elementwise_layer 1 - auto ew1_layer = add_elementwise(engine, nvinfer1::ElementWiseOperation::kPOW, unary_output, p, - layer_info(node) + "_pow_for_unary"); - - POROS_CHECK(ew1_layer, "Unable to create POW layer from node: " << *node); - auto ew_output = ew1_layer->getOutput(0); - - // reduce_layer - auto reduce_layer = engine->network()->addReduce(*ew_output, nvinfer1::ReduceOperation::kSUM, dim, keepdim); - reduce_layer->setName((layer_info(node) + "_IReduceLayer").c_str()); - auto reduce_output = reduce_layer->getOutput(0); - - // elementwise_layer 2 - auto ew2_layer = add_elementwise(engine, nvinfer1::ElementWiseOperation::kPOW, reduce_output, p_inverse, - layer_info(node) + "_pow_for_reduce"); - - engine->context().set_tensor(node->outputs()[0], ew2_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << ew2_layer->getOutput(0)->getDimensions(); - return true; -} - -// -POROS_REGISTER_CONVERTER(TensorrtEngine, NormConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, FrobeniusNormConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/norm.h b/poros/poros/converter/gpu/norm.h deleted file mode 100644 index 3da51ead59e..00000000000 --- a/poros/poros/converter/gpu/norm.h +++ /dev/null @@ -1,98 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file norm.h -* @author Lin Xiao Chun (linxiaochun@baidu.com) -* @date 2022-02-23 20:33:45 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class NormConverter : public GpuConverter { -public: - NormConverter() {} - - virtual ~NormConverter() {} - - bool converter(TensorrtEngine *engine, const torch::jit::Node *node); - - //aten::norm.ScalarOpt_dim(Tensor self, Scalar? p, int[1] dim, bool keepdim=False) -> Tensor - const std::vector schema_string() { - return { - "aten::norm.ScalarOpt_dim(Tensor self, Scalar? p, int[1] dim, bool keepdim=False) -> Tensor", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * - * **/ - const std::vector node_kind() { - return {torch::jit::aten::norm}; - } - - bool assign_schema_attr() { - bool result = true; - result &= assign_schema_attr_helper({{"aten::norm.ScalarOpt_dim(Tensor self, Scalar? p, int[1] dim, bool keepdim=False) -> Tensor", {1, 1}}}); - return result; - } - -}; - -class FrobeniusNormConverter : public GpuConverter { -public: - FrobeniusNormConverter() {} - - virtual ~FrobeniusNormConverter() {} - - bool converter(TensorrtEngine *engine, const torch::jit::Node *node); - - //aten::frobenius_norm.dim(Tensor self, int[1] dim, bool keepdim=False) -> Tensor - const std::vector schema_string() { - return { - "aten::frobenius_norm.dim(Tensor self, int[1] dim, bool keepdim=False) -> Tensor", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * - * **/ - const std::vector node_kind() { - return {torch::jit::aten::frobenius_norm}; - } - - bool assign_schema_attr() { - bool result = true; - result &= assign_schema_attr_helper({{"aten::frobenius_norm.dim(Tensor self, int[1] dim, bool keepdim=False) -> Tensor", {1, 1}}}); - return result; - } - -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/plugins/interpolate_plugin.cpp b/poros/poros/converter/gpu/plugins/interpolate_plugin.cpp deleted file mode 100644 index ef206805930..00000000000 --- a/poros/poros/converter/gpu/plugins/interpolate_plugin.cpp +++ /dev/null @@ -1,398 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file interpolate_plugin.cpp -* @author tianjinjin@baidu.com -* @date Mon Sep 27 16:18:38 CST 2021 -* @brief -**/ - -#include "torch/torch.h" - -#include "poros/converter/gpu/plugins/interpolate_plugin.h" -#include "poros/engine/trtengine_util.h" -#include "poros/util/macros.h" - -namespace baidu { -namespace mirana { -namespace poros { - -InterpolatePlugin::InterpolatePlugin(std::vector in_shape, - std::vector out_shape, - std::vector size, - std::vector scales, - std::string mode, - bool align_corners, - bool use_scales) - : in_shape_(in_shape), out_shape_(out_shape), size_(size), - scales_(scales), mode_(mode), align_corners_(align_corners), use_scales_(use_scales) { - if (use_scales) { - POROS_ASSERT(mode_ != "adaptive_avg_pool2d", - "use_scales is not valid for adaptive_avg_pool2d"); - POROS_ASSERT(scales_.size() != 0, - "Attempted to use interpolate plugin without providing scales while use_scales=true"); - at::Tensor input = at::randint(1, 10, in_shape, {at::kCUDA}); - at::Tensor output; - - if (mode_ == "linear") { - output = at::upsample_linear1d(input, c10::nullopt, align_corners_, scales_[0]); - } else if (mode_ == "bilinear") { - output = at::upsample_bilinear2d(input, c10::nullopt, align_corners_, scales_); - std::cout << output.sizes() << std::endl; - } else if (mode_ == "trilinear") { - output = at::upsample_trilinear3d(input, c10::nullopt, align_corners_, scales_); - } - - out_shape_ = output.sizes().vec(); - } else { - POROS_ASSERT((size_.size() != 0 && out_shape_.size() != 0), - "Attempted to use interpolate plugin without providing output size while use_scales=false"); - } -} - -InterpolatePlugin::InterpolatePlugin(const char* data, size_t length) { - std::istringstream data_stream(std::string(data, length)); - - torch::serialize::InputArchive input_archive; - input_archive.load_from(data_stream); - { - torch::IValue value; - input_archive.read("in_shape", value); - in_shape_ = value.toIntVector(); - } - { - torch::IValue value; - input_archive.read("out_shape", value); - out_shape_ = value.toIntVector(); - } - { - torch::IValue value; - input_archive.read("size", value); - size_ = value.toIntVector(); - } - { - torch::IValue value; - input_archive.read("scales", value); - scales_ = value.toDoubleVector(); - } - { - torch::IValue value; - input_archive.read("mode", value); - mode_ = value.toStringRef(); - } - { - torch::IValue value; - input_archive.read("align_corners", value); - align_corners_ = value.toBool(); - } - { - torch::IValue value; - input_archive.read("use_scales", value); - use_scales_ = value.toBool(); - } -} - -std::vector InterpolatePlugin::getInputShape() { - return in_shape_; -} - -std::vector InterpolatePlugin::getOutputShape() { - return out_shape_; -} - -std::vector InterpolatePlugin::getOutputSize() { - return size_; -} - -int InterpolatePlugin::getNbOutputs() const noexcept { - if (mode_ == "adaptive_max_pool2d") { - return 2; - } else { - return 1; - } -} - -const char* InterpolatePlugin::getPluginType() const noexcept { - return "Interpolate"; -} - -const char* InterpolatePlugin::getPluginVersion() const noexcept { - return "1"; -} - -const char* InterpolatePlugin::getPluginNamespace() const noexcept { - return ""; -} - -nvinfer1::IPluginV2DynamicExt* InterpolatePlugin::clone() const noexcept { - return new InterpolatePlugin(in_shape_, out_shape_, size_, scales_, mode_, align_corners_, use_scales_); -} - -nvinfer1::DimsExprs InterpolatePlugin::getOutputDimensions(int outputIndex, - const nvinfer1::DimsExprs* inputs, - int nbInputs, - nvinfer1::IExprBuilder& exprBuilder) noexcept { - nvinfer1::DimsExprs output(inputs[0]); - - // TODO: This should enable the case of using this plugin with dynamic shape, scale factor and align corners == true - // to cover the different implementations between PyTorch and TRT. However TRT currently does not support doubles for - // ExprBuilder constants. Once that is possible enable this code and remove the code in the constructor if - // if (use_scales_) { - // auto input_dimsexprs = inputs[0]; - // output.d[0] = exprBuilder.operation(DimensionOperation::kMAX, *input_dimsexprs.d[0], *exprBuilder.constant(0)); - // if (mode_ == "linear") { - // output.d[1] = exprBuilder.operation(DimensionOperation::kPROD, *input_dimsexprs.d[1], - // *exprBuilder.constant(scales_[1])); - // } else if (mode_ == "bilinear") { - // output.d[1] = exprBuilder.operation(DimensionOperation::kPROD, *input_dimsexprs.d[1], - // *exprBuilder.constant(scales_[1])); output.d[2] = exprBuilder.operation(DimensionOperation::kPROD, - // *input_dimsexprs.d[2], *exprBuilder.constant(scales_[2])); - // } else if (mode_ == "trilinear") { - // output.d[1] = exprBuilder.operation(DimensionOperation::kPROD, *input_dimsexprs.d[1], - // *exprBuilder.constant(scales_[1])); output.d[2] = exprBuilder.operation(DimensionOperation::kPROD, - // *input_dimsexprs.d[2], *exprBuilder.constant(scales_[2])); output.d[3] = - // exprBuilder.operation(DimensionOperation::kPROD, *input_dimsexprs.d[3], *exprBuilder.constant(scales_[3])); - // } - // } else { - for (unsigned int i = 0; i < out_shape_.size(); i++) { - output.d[i] = exprBuilder.constant(out_shape_[i]); - } - //} - return output; -} - -nvinfer1::DataType InterpolatePlugin::getOutputDataType(int index, - const nvinfer1::DataType* inputTypes, - int nbInputs) const noexcept { - return nvinfer1::DataType::kFLOAT; -} - -int InterpolatePlugin::initialize() noexcept { - return 0; -} - -void InterpolatePlugin::serialize(void* buffer) const noexcept { - std::string data = serializeToString(); - size_t size = getSerializationSize(); - data.copy((char*)buffer, size); -} - -std::string InterpolatePlugin::serializeToString() const { - torch::serialize::OutputArchive output_archive; - - output_archive.write("in_shape", torch::IValue(in_shape_)); - output_archive.write("out_shape", torch::IValue(out_shape_)); - output_archive.write("size", torch::IValue(size_)); - output_archive.write("scales", torch::IValue(scales_)); - output_archive.write("mode", torch::IValue(mode_)); - output_archive.write("align_corners", torch::IValue(align_corners_)); - output_archive.write("use_scales", torch::IValue(use_scales_)); - - std::ostringstream data_str; - output_archive.save_to(data_str); - - return data_str.str(); -} - -size_t InterpolatePlugin::getSerializationSize() const noexcept { - return serializeToString().size(); -} - -bool InterpolatePlugin::supportsFormatCombination(int pos, - const nvinfer1::PluginTensorDesc* inOut, - int nbInputs, - int nbOutputs) noexcept { - if (nbInputs != 1) { - LOG(WARNING) << "Expected a single tensor as input to interpolate plugin"; - } - if (mode_ == "adaptive_max_pool2d") { - if (nbOutputs != 2) { - LOG(WARNING) << "Expected 2 tensors as output to interpolate plugin"; - } - if (pos < 0 || pos > 2) { - LOG(WARNING) << "There should be exactly 3 connections to the plugin - 1 input, 2 output"; - } - } else { - if (nbOutputs != 1) { - LOG(WARNING) << "Expected a single tensor as output to interpolate plugin"; - } - if (pos < 0 || pos > 1) { - LOG(WARNING) << "There should be exactly 2 connections to the plugin - 1 input, 1 output"; - } - } - - const nvinfer1::PluginTensorDesc& in = inOut[0]; - - if (pos == 0) { - return (in.type == nvinfer1::DataType::kFLOAT) && (in.format == nvinfer1::TensorFormat::kLINEAR); - } - - // pos == 1, accessing information about output tensor - const nvinfer1::PluginTensorDesc& out = inOut[1]; - - return (in.type == out.type) && (in.format == out.format); -} - -void InterpolatePlugin::configurePlugin(const nvinfer1::DynamicPluginTensorDesc* in, - int nbInputs, - const nvinfer1::DynamicPluginTensorDesc* out, - int nbOutputs) noexcept { - dtype_ = nvinfer1::DataType::kFLOAT; -} - -size_t InterpolatePlugin::getWorkspaceSize(const nvinfer1::PluginTensorDesc* inputs, - int nbInputs, - const nvinfer1::PluginTensorDesc* outputs, - int nbOutputs) const noexcept { - return 0; -} - -int InterpolatePlugin::enqueue(const nvinfer1::PluginTensorDesc* inputDesc, - const nvinfer1::PluginTensorDesc* outputDesc, - const void* const* inputs, - void* const* outputs, - void* workspace, - cudaStream_t stream) noexcept { - at::Tensor input = - at::from_blob((void*)inputs[0], nvdim_to_sizes(inputDesc->dims), [](void*) {}, {at::kCUDA}).to(torch::kFloat); - at::Tensor output = - at::from_blob(outputs[0], nvdim_to_sizes(outputDesc->dims), [](void*) {}, {at::kCUDA}).to(torch::kFloat); - - at::cuda::CUDAStream torch_stream = at::cuda::getStreamFromPool(); - at::cuda::CUDAStreamGuard torch_guard(torch_stream); - - cudaEvent_t event; - cudaEventCreate(&event); - cudaEventRecord(event, stream); - - cudaStreamWaitEvent(torch_stream.stream(), event, 0); - at::Tensor out; - if (use_scales_) { - if (mode_ == "linear") { - out = at::upsample_linear1d(input, c10::nullopt, align_corners_, {scales_[0]}); - } else if (mode_ == "bilinear") { - out = at::upsample_bilinear2d(input, c10::nullopt, align_corners_, scales_); - } else if (mode_ == "trilinear") { - out = at::upsample_trilinear3d(input, c10::nullopt, align_corners_, scales_); - } - } else { - if (mode_ == "linear") { - out = at::upsample_linear1d(input, {size_[0]}, align_corners_); - } else if (mode_ == "bilinear") { - out = at::upsample_bilinear2d(input, {size_[0], size_[1]}, align_corners_); - } else if (mode_ == "trilinear") { - out = at::upsample_trilinear3d(input, {size_[0], size_[1], size_[2]}, align_corners_); - } else if (mode_ == "adaptive_avg_pool2d") { - out = at::adaptive_avg_pool2d(input, {size_[0], size_[1]}); - } else if (mode_ == "adaptive_max_pool2d") { - out = std::get<0>(at::adaptive_max_pool2d(input, {size_[0], size_[1]})); - } - } - - output.copy_(out); - cudaEvent_t torch_event; - cudaEventCreate(&torch_event); - cudaEventRecord(torch_event, torch_stream.stream()); - - cudaStreamWaitEvent(stream, torch_event, 0); - - cudaEventDestroy(event); - cudaEventDestroy(torch_event); - - return 0; -} - -/* - * InterpolatePluginCreator class implementations - */ - -InterpolatePluginCreator::InterpolatePluginCreator() { - mPluginAttributes.emplace_back(nvinfer1::PluginField("in_shape", nullptr, nvinfer1::PluginFieldType::kINT32, 1)); - mPluginAttributes.emplace_back(nvinfer1::PluginField("out_shape", nullptr, nvinfer1::PluginFieldType::kINT32, 1)); - mPluginAttributes.emplace_back(nvinfer1::PluginField("out_size", nullptr, nvinfer1::PluginFieldType::kINT32, 1)); - mPluginAttributes.emplace_back(nvinfer1::PluginField("scales", nullptr, nvinfer1::PluginFieldType::kFLOAT32, 1)); - mPluginAttributes.emplace_back(nvinfer1::PluginField("mode", nullptr, nvinfer1::PluginFieldType::kCHAR, 1)); - mPluginAttributes.emplace_back(nvinfer1::PluginField("align_corners", nullptr, nvinfer1::PluginFieldType::kINT32, 1)); - mPluginAttributes.emplace_back(nvinfer1::PluginField("use_scales", nullptr, nvinfer1::PluginFieldType::kINT32, 1)); - - mFC.nbFields = mPluginAttributes.size(); - mFC.fields = mPluginAttributes.data(); -} - -const char* InterpolatePluginCreator::getPluginNamespace() const noexcept { - return ""; -} - -const char* InterpolatePluginCreator::getPluginName() const noexcept { - return "Interpolate"; -} - -const char* InterpolatePluginCreator::getPluginVersion() const noexcept { - return "1"; -} - -nvinfer1::IPluginV2* InterpolatePluginCreator::createPlugin(const char* name, - const nvinfer1::PluginFieldCollection* fc) noexcept { - std::vector in_shape; - std::vector out_shape; - std::vector out_size; - std::vector scales; - std::string mode; - int32_t align_corners = 0; - int32_t use_scales = 0; - - for (int i = 0; i < fc->nbFields; i++) { - std::string field_name(fc->fields[i].name); - if (field_name.compare("in_shape") == 0) { - auto in_shape_values = static_cast(fc->fields[i].data); - in_shape.assign(in_shape_values, in_shape_values + fc->fields[i].length); - } else if (field_name.compare("out_shape") == 0) { - auto out_shape_values = static_cast(fc->fields[i].data); - out_shape.assign(out_shape_values, out_shape_values + fc->fields[i].length); - } else if (field_name.compare("out_size") == 0) { - auto out_size_values = static_cast(fc->fields[i].data); - out_size.assign(out_size_values, out_size_values + fc->fields[i].length); - } else if (field_name.compare("scales") == 0) { - auto scales_values = static_cast(fc->fields[i].data); - scales.assign(scales_values, scales_values + fc->fields[i].length); - } else if (field_name.compare("mode") == 0) { - mode = *static_cast(fc->fields[i].data); - } else if (field_name.compare("align_corners") == 0) { - align_corners = *static_cast(fc->fields[i].data); - } else if (field_name.compare("use_scales") == 0) { - use_scales = *static_cast(fc->fields[i].data); - } - } - InterpolatePlugin* plugin = - new InterpolatePlugin(in_shape, out_shape, out_size, scales, mode, (bool)align_corners, (bool)use_scales); - return plugin; -} - -nvinfer1::IPluginV2* InterpolatePluginCreator::deserializePlugin(const char* name, - const void* serialData, - size_t serialLength) noexcept { - name_ = name; - return new InterpolatePlugin((const char*)serialData, serialLength); -} - -const nvinfer1::PluginFieldCollection* InterpolatePluginCreator::getFieldNames() noexcept { - return nullptr; -} - -REGISTER_TENSORRT_PLUGIN(InterpolatePluginCreator); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/plugins/interpolate_plugin.h b/poros/poros/converter/gpu/plugins/interpolate_plugin.h deleted file mode 100644 index 05a68989725..00000000000 --- a/poros/poros/converter/gpu/plugins/interpolate_plugin.h +++ /dev/null @@ -1,289 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file interpolate_plugin.h -* @author tianjinjin@baidu.com -* @date Mon Sep 27 16:18:38 CST 2021 -* @brief -**/ - -#pragma once - -#include -#include -#include -#include -#include -#include -#include -#include - -//from tensorrt -#include "NvInferPlugin.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/* -* @brief InterpolatePlugin 实现tensorrt的一个插件。 - 该插件支持的pooling类算子包括:adaptive_avg_pool2d & adaptive_max_pool2d - 该插件支持的linear类算子包括:linear & bilinear & trilinear - 该插件会被上述算子对应的gpu-converter调用,注册到gpu-engine上,实现gpu-engine对该算子的支持; - plugin的内部逻辑,主要是调用了pytorch-aten的实现(CUDA实现)。 -**/ -class InterpolatePlugin : public nvinfer1::IPluginV2DynamicExt { -public: - /* - * @brief 多参数构造函数。 - * @param [in] in_shape : 输入tensors的shape信息 - * @param [in] out_shape : 输出tensors的shape信息 - * @param [in] size : 输出的 spatial 尺寸(当mode=linear, bilinear 和 trilinear) - * 输出的目标 size 信息(当mode=adaptive_avg_pool2d 和 adaptive_max_pool2d) - * @param [in] scales : spatial 尺寸的缩放因子, 当use_scales=True时生效。 - * @param [in] mode : 算子标识,可选adaptive_avg_pool2d、adaptive_max_pool2d、linear、bilinear、trilinear。 - * @param [in] align_corners : 默认为false,如果 align_corners=True,则对齐input和output 的角点像素(corner pixels),保持在角点像素的值. - * 只会对 mode=linear, bilinear 和 trilinear 有作用。 - * @param [in] use_scales : 默认为false,如果use_scales=True,入参scales生效(且scales必须不为空)。 - * 只会对 mode=linear, bilinear 和 trilinear 有作用。 - **/ - InterpolatePlugin( - std::vector in_shape, - std::vector out_shape, - std::vector size, - std::vector scales, - std::string mode, - bool align_corners, - bool use_scales); - - /* - * @brief deserialize阶段使用的构造函数。 - * @param [in] data : 序列化好的data。 - * @param [in] length : 数据长度。 - **/ - InterpolatePlugin(const char* data, size_t length); - - /* - * @brief InterpolatePlugin不应该存在无参数构造函数,将该默认构造函数删除。 - **/ - InterpolatePlugin() = delete; - - /***************************************************************************** - 以下部分是tensorrt定义的 IPluginV2DynamicExt API - ******************************************************************************/ - /* - * @brief clone函数,将这个插件对象克隆给tensorrt的builder/network/engine。 - **/ - nvinfer1::IPluginV2DynamicExt* clone() const noexcept override; - - /* - * @brief 返回输出维度信息。 - **/ - nvinfer1::DimsExprs getOutputDimensions( - int outputIndex, - const nvinfer1::DimsExprs* inputs, - int nbInputs, - nvinfer1::IExprBuilder& exprBuilder) noexcept override; - - /* - * @brief 插件的输入输出是否支持inOut[pos].format和inOut[pos].type指定的格式/数据类型。 - **/ - bool supportsFormatCombination( - int pos, - const nvinfer1::PluginTensorDesc* inOut, - int nbInputs, - int nbOutputs) noexcept override; - - /* - * @brief 配置这个插件,判断输入和输出类型数量是否正确等。 - **/ - void configurePlugin( - const nvinfer1::DynamicPluginTensorDesc* in, - int nbInputs, - const nvinfer1::DynamicPluginTensorDesc* out, - int nbOutputs) noexcept override; - /* - * @brief 返回需要中间显存变量的实际数据大小(bytesize)。 - **/ - size_t getWorkspaceSize( - const nvinfer1::PluginTensorDesc* inputs, - int nbInputs, - const nvinfer1::PluginTensorDesc* outputs, - int nbOutputs) const noexcept override; - /* - * @brief 该插件的的实际执行函数(重要!)。 - **/ - int enqueue( - const nvinfer1::PluginTensorDesc* inputDesc, - const nvinfer1::PluginTensorDesc* outputDesc, - const void* const* inputs, - void* const* outputs, - void* workspace, - cudaStream_t stream) noexcept override; - - /***************************************************************************** - 以下部分是tensorrt定义的 IPluginV2Ext API - ******************************************************************************/ - /* - * @brief 返回结果的数据类型(一般与输入类型一致)。 - **/ - nvinfer1::DataType getOutputDataType(int index, const nvinfer1::DataType* inputTypes, int nbInputs) const noexcept override; - - /***************************************************************************** - 以下部分是tensorrt定义的 IPluginV2 API - ******************************************************************************/ - /* - * @brief 插件名称信息。 - **/ - const char* getPluginType() const noexcept override; - /* - * @brief 插件版本信息。 - **/ - const char* getPluginVersion() const noexcept override; - /* - * @brief 插件返回多少个Tensor。 - **/ - int getNbOutputs() const noexcept override; - - /* - * @brief 初始化函数,在这个插件准备开始run之前执行。 - **/ - int initialize() noexcept override; - - /* - * @brief 资源释放函数,engine destory的时候调用该函数。 - **/ - void terminate() noexcept override {} - - /* - * @brief 返回序列化时需要写多少字节到buffer中。 - **/ - size_t getSerializationSize() const noexcept override; - - /* - * @brief 把需要用的数据按照顺序序列化到buffer中。 - **/ - void serialize(void* buffer) const noexcept override; - - /* - * @brief 销毁插件对象,network/builder/engine destroy的时候调用该函数。 - **/ - void destroy() noexcept override {} - - /* - * @brief 设置插件的namespace,默认为 ”“。 - **/ - void setPluginNamespace(const char* pluginNamespace) noexcept override {}; - - /* - * @brief 返回插件的namespace。 - **/ - const char* getPluginNamespace() const noexcept override; - - /***************************************************************************** - 以下部分为本插件自定义的API - ******************************************************************************/ - /* - * @brief 返回输入的shape信息。 - **/ - std::vector getInputShape(); - - /* - * @brief 返回输出的shape信息。 - **/ - std::vector getOutputShape(); - - /* - * @brief 返回输出的spatial尺寸。 - **/ - std::vector getOutputSize(); - - /* - * @brief 插件序列化函数。 - **/ - std::string serializeToString() const; - -private: - nvinfer1::DataType dtype_; - - std::vector in_shape_; - std::vector out_shape_; - std::vector size_; - std::vector scales_; - std::string mode_; - bool align_corners_; - bool use_scales_; -}; - -/* -* @brief InterpolatePluginCreator 实现插件的注册。 -* (配合全局宏REGISTER_TENSORRT_PLUGIN,将插件注册到tensorrt, 此后插件可以通过getPluginRegistry获取) -**/ -class InterpolatePluginCreator : public nvinfer1::IPluginCreator { -public: - /* - * @brief 默认构造函数。 - **/ - InterpolatePluginCreator(); - - /* - * @brief 获取插件的名称信息。 - **/ - const char* getPluginName() const noexcept override; - - /* - * @brief 获取插件的版本信息。 - **/ - const char* getPluginVersion() const noexcept override; - - /* - * @brief 获取创建该插件需要的字段列表,该列表被 createPlugin 使用。 - **/ - const nvinfer1::PluginFieldCollection* getFieldNames() noexcept override; - - /* - * @brief 根据插件名和字段列表信息,创建插件对象。 - **/ - nvinfer1::IPluginV2* createPlugin( - const char* name, - const nvinfer1::PluginFieldCollection* fc) noexcept override; - - /* - * @brief 反序列化插件。 - **/ - nvinfer1::IPluginV2* deserializePlugin( - const char* name, - const void* serialData, - size_t serialLength) noexcept override; - - /* - * @brief 设置插件的namespace信息。 - **/ - void setPluginNamespace(const char* libNamespace) noexcept override{}; - - /* - * @brief 获取插件的namespace信息。 - **/ - const char* getPluginNamespace() const noexcept override; - -private: - std::string name_; - std::vector mPluginAttributes; - nvinfer1::PluginFieldCollection mFC; -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/pooling.cpp b/poros/poros/converter/gpu/pooling.cpp deleted file mode 100644 index ece7ad65390..00000000000 --- a/poros/poros/converter/gpu/pooling.cpp +++ /dev/null @@ -1,266 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file pooling.cpp -* @author tianjinjin@baidu.com -* @date Wed Aug 18 11:25:13 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/converter_util.h" -#include "poros/converter/gpu/pooling.h" -#include "poros/converter/gpu/weight.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -//note1: max_pool?d 输入参数都是6个,各个参数的含义是一致的,差异在于 int[] 的维度不一样。 -//note2: avg_pool1d 的输入参数是6个,avg_pool2d 和 3d 的输入参数是7个,多了最后一个参数 divisor_override。 -//note3: max 与 avg 从第5个参数开始,出现了定义上的差异,需要注意。 -bool PoolingConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 6 || inputs.size() == 7), - "invaid inputs size for PoolingConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for PoolingConverter is not Tensor as expected"); - - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - - // Max Pool needs at least 4D input - auto orig_dims = in->getDimensions(); - bool expandDims = (orig_dims.nbDims < 4); - if (expandDims) { - in = add_padding(engine, node, in, 4, false, true); - } - - auto kernel_size = sizes_to_nvdim((engine->context().get_constant(inputs[1])).toIntList()); - auto stride = sizes_to_nvdim((engine->context().get_constant(inputs[2])).toIntList()); - auto padding = sizes_to_nvdim((engine->context().get_constant(inputs[3])).toIntList()); - if (stride.nbDims == 0) { - LOG(INFO) << "Stride not provided, using kernel_size as stride"; - stride = sizes_to_nvdim((engine->context().get_constant(inputs[1])).toIntList()); - } - if (kernel_size.nbDims == 1) { - kernel_size = unsqueeze_dims(kernel_size, 0, 1); - } - if (padding.nbDims == 1) { - padding = unsqueeze_dims(padding, 0, 0); - } - if (stride.nbDims == 1) { - stride = unsqueeze_dims(stride, 0, 1); - } - LOG(INFO) << "kernel_size: " << kernel_size << ", padding: " << padding << ", stride: " << stride; - - - bool ceil_mode = false; - nvinfer1::IPoolingLayer* new_layer; - - //when it's max pooling - if (node->kind() == torch::jit::aten::max_pool1d || - node->kind() == torch::jit::aten::max_pool2d || - node->kind() == torch::jit::aten::max_pool3d) { - auto dilation = sizes_to_nvdim((engine->context().get_constant(inputs[4])).toIntList()); - POROS_CHECK(dilation == sizes_to_nvdim(std::vector(dilation.nbDims, 1)), - "Pooling dilation is not supported in TensorRT"); - - LOG(INFO) << "dilation: " << dilation; - LOG(WARNING) << "Dilation not used in Max pooling converter"; - - ceil_mode = (engine->context().get_constant(inputs[5])).toBool(); - - new_layer = engine->network()->addPoolingNd(*in, nvinfer1::PoolingType::kMAX, kernel_size); - POROS_CHECK(new_layer, "Unable to create Max Pooling layer from node: " << *node); - new_layer->setName((layer_info(node) + "_IPoolingLayer_max").c_str()); - - //when it's avg pooling - } else if (node->kind() == torch::jit::aten::avg_pool1d || - node->kind() == torch::jit::aten::avg_pool2d || - node->kind() == torch::jit::aten::avg_pool3d) { - - ceil_mode = (engine->context().get_constant(inputs[4])).toBool(); - bool count_inlcude_pad = (engine->context().get_constant(inputs[5])).toBool(); - - new_layer = engine->network()->addPoolingNd(*in, nvinfer1::PoolingType::kAVERAGE, kernel_size); - POROS_CHECK(new_layer, "Unable to create Avg Pooling layer from node: " << *node); - new_layer->setAverageCountExcludesPadding(!count_inlcude_pad); - new_layer->setName((layer_info(node) + "_IPoolingLayer_average").c_str()); - - //we should never reach here - } else { - POROS_THROW_ERROR("Unsupported pool mode!"); - } - - auto padding_mode = - ceil_mode ? nvinfer1::PaddingMode::kEXPLICIT_ROUND_UP : nvinfer1::PaddingMode::kEXPLICIT_ROUND_DOWN; - - new_layer->setPaddingMode(padding_mode); - new_layer->setPaddingNd(padding); - new_layer->setStrideNd(stride); - - auto out_tensor = add_unpadding(engine, node, new_layer->getOutput(0), orig_dims.nbDims, false, true); - - // avg_pool2d or avg_pool3d divisor_override - if (node->kind() == torch::jit::aten::avg_pool2d || - node->kind() == torch::jit::aten::avg_pool3d) { - auto maybe_divisor = engine->context().get_constant(inputs[6]); - if (maybe_divisor.isScalar()) { - auto divisor = maybe_divisor.toScalar().to(); - if (divisor != 0){ - auto kernel_size_list = sizes_to_nvdim((engine->context().get_constant(inputs[1])).toIntList()); - int64_t kernel_area = 1; - for(auto i = 0; i < kernel_size_list.nbDims; i++) { - kernel_area *= kernel_size_list.d[i]; - } - auto actual_divisor = tensor_to_const(engine, torch::tensor({(float)kernel_area / (float)divisor})); - auto mul = add_elementwise(engine, nvinfer1::ElementWiseOperation::kPROD, - out_tensor, actual_divisor, layer_info(node) + "_prod"); - POROS_CHECK(mul, "Unable to create mul layer from node: " << *node); - out_tensor = mul->getOutput(0); - } else { - LOG(INFO) << "Invalid parameter: divisor_override"; - return false; - } - } - } - - engine->context().set_tensor(node->outputs()[0], out_tensor); - LOG(INFO) << "Output tensor shape: " << out_tensor->getDimensions(); - return true; -} - -//note1: adaptive_avg_pool?d,各个参数的含义是一致的,差异在于 int[] 的维度不一样。 -//note2: adaptive_max_pool?d,同note1, 各个参数含义一样,差异在于 int[] 的维度不一样。 -//note3: avg 与 max的 输入参数都是2个,avg的输出参数是1个,max的输出参数是2个。 -//note4: 这个家族的6个op,没有全部实现,需要注意!!! -bool AdaptivePoolingConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for AdaptivePoolingConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for AdaptivePoolingConverter is not Tensor as expected"); - - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - nvinfer1::Dims orig_dims = in->getDimensions(); - auto out_size = sizes_to_nvdim((engine->context().get_constant(inputs[1])).toIntList()); - LOG(INFO) << "get out_size: " << out_size << " in AdaptivePoolingConverter"; - - nvinfer1::PoolingType pool_type; - if (node->kind() == torch::jit::aten::adaptive_avg_pool1d || - node->kind() == torch::jit::aten::adaptive_avg_pool2d) { - pool_type = nvinfer1::PoolingType::kAVERAGE; - } else if (node->kind() == torch::jit::aten::adaptive_max_pool2d) { - pool_type = nvinfer1::PoolingType::kMAX; - } else { - POROS_THROW_ERROR("Unsupported Adaptive pool mode!"); - } - - // Corner case: when out dimension is all ones, replace with simpler operation - if (out_size.d[0] == 1 && (out_size.nbDims < 2 || out_size.d[1] == 1) && - (out_size.nbDims < 3 || out_size.d[2] == 1)) { - LOG(INFO) << "Matched corner case in AdaptivePoolingConverter"; - // Generate a bitmask of all 1s except the last 2 bits (N and C axes) - uint32_t reduceAxes = ((1 << orig_dims.nbDims) - 1) & ~0b11; - auto* new_layer = engine->network()->addReduce( - *in, - pool_type == nvinfer1::PoolingType::kMAX ? nvinfer1::ReduceOperation::kMAX : nvinfer1::ReduceOperation::kAVG, - reduceAxes, - /*keepDimensions=*/true); - new_layer->setName((layer_info(node) + "_IReduceLayer").c_str()); - - engine->context().set_tensor(node->outputs()[0], new_layer->getOutput(0)); - LOG(INFO) << "AdaptivePoolingConverter: Output tensor shape: " << new_layer->getOutput(0)->getDimensions(); - return true; - } - - bool expandDims = (orig_dims.nbDims < 4); - POROS_CHECK(orig_dims.nbDims > 2, "Unable to create pooling layer from node: " << *node); - if (expandDims) { - in = add_padding(engine, node, in, 4, false, false); - } - - if (out_size.nbDims == 1) { - out_size = unsqueeze_dims(out_size, 0, 1); - } - - auto in_shape = nvdim_to_sizes(in->getDimensions()); - nvinfer1::ILayer* new_layer = nullptr; - - nvinfer1::PluginFieldCollection fc; - std::vector f; - auto out_shape = in_shape; - auto out_size_vec = nvdim_to_sizes(out_size); - - std::copy(out_size_vec.begin(), out_size_vec.end(), out_shape.begin() + (in_shape.size() - out_size_vec.size())); - std::vector in_shape_casted(in_shape.begin(), in_shape.end()); - f.emplace_back( - nvinfer1::PluginField("in_shape", in_shape_casted.data(), nvinfer1::PluginFieldType::kINT32, in_shape.size())); - std::vector out_shape_casted(out_shape.begin(), out_shape.end()); - f.emplace_back( - nvinfer1::PluginField("out_shape", out_shape_casted.data(), nvinfer1::PluginFieldType::kINT32, out_shape.size())); - std::vector out_size_casted(out_size_vec.begin(), out_size_vec.end()); - f.emplace_back( - nvinfer1::PluginField("out_size", out_size_casted.data(), nvinfer1::PluginFieldType::kINT32, out_size_vec.size())); - f.emplace_back( - nvinfer1::PluginField("scales", nullptr, nvinfer1::PluginFieldType::kFLOAT64, 0)); - - int32_t align_corners_casted = 0; - f.emplace_back( - nvinfer1::PluginField("align_corners", &align_corners_casted, nvinfer1::PluginFieldType::kINT32, 1)); - int32_t use_scales_casted = 0; - f.emplace_back( - nvinfer1::PluginField("use_scales", &use_scales_casted, nvinfer1::PluginFieldType::kINT32, 1)); - - std::string mode = "adaptive_avg_pool2d"; - if (pool_type == nvinfer1::PoolingType::kMAX) { - mode = "adaptive_max_pool2d"; - } - f.emplace_back( - nvinfer1::PluginField("mode", &mode, nvinfer1::PluginFieldType::kCHAR, 1)); - - fc.nbFields = f.size(); - fc.fields = f.data(); - - auto creator = getPluginRegistry()->getPluginCreator("Interpolate", "1", ""); - auto interpolate_plugin = creator->createPlugin(mode.c_str(), &fc); - LOG(INFO) << "create Interpolate plugin done"; - - new_layer = engine->network()->addPluginV2(reinterpret_cast(&in), 1, *interpolate_plugin); - POROS_CHECK(new_layer, "Unable to create pooling (interpolation) plugin from node" << *node); - new_layer->setName((layer_info(node) + "_plugin_Interpolate").c_str()); - auto layer_output = add_unpadding(engine, node, new_layer->getOutput(0), orig_dims.nbDims, false, false); - - engine->context().set_tensor(node->outputs()[0], layer_output); - LOG(INFO) << "Output tensor shape: " << layer_output->getDimensions(); - //attention: 对于adaptive_max_pool2d, 映射第二个output - if (mode == "adaptive_max_pool2d") { - engine->context().set_tensor(node->outputs()[1], new_layer->getOutput(1)); - LOG(INFO) << "Output tensor2 shape: " << new_layer->getOutput(1)->getDimensions(); - } - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, PoolingConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, AdaptivePoolingConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/converter/gpu/pooling.h b/poros/poros/converter/gpu/pooling.h deleted file mode 100644 index 45a628f2784..00000000000 --- a/poros/poros/converter/gpu/pooling.h +++ /dev/null @@ -1,96 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file pooling.h -* @author tianjinjin@baidu.com -* @date Tue Aug 17 22:57:03 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class PoolingConverter : public GpuConverter { -public: - PoolingConverter() {} - virtual ~PoolingConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::max_pool1d(Tensor self, int[1] kernel_size, int[1] stride=[], int[1] padding=0, int[1] dilation=1, bool ceil_mode=False) -> Tensor", - "aten::max_pool2d(Tensor self, int[2] kernel_size, int[2] stride=[], int[2] padding=0, int[2] dilation=1, bool ceil_mode=False) -> Tensor", - "aten::max_pool3d(Tensor self, int[3] kernel_size, int[3] stride=[], int[3] padding=0, int[3] dilation=1, bool ceil_mode=False) -> Tensor", - "aten::avg_pool1d(Tensor self, int[1] kernel_size, int[1] stride=[], int[1] padding=0, bool ceil_mode=False, bool count_include_pad=True) -> Tensor", - "aten::avg_pool2d(Tensor self, int[2] kernel_size, int[2] stride=[], int[2] padding=0, bool ceil_mode=False, bool count_include_pad=True, int? divisor_override=None) -> Tensor", - "aten::avg_pool3d(Tensor self, int[3] kernel_size, int[3] stride=[], int[3] padding=0, bool ceil_mode=False, bool count_include_pad=True, int? divisor_override=None) -> Tensor" - }; - } - - const std::vector node_kind() { - return {torch::jit::aten::max_pool1d, - torch::jit::aten::avg_pool1d, - torch::jit::aten::max_pool2d, - torch::jit::aten::avg_pool2d, - torch::jit::aten::max_pool3d, - torch::jit::aten::avg_pool3d}; - } -}; - - -class AdaptivePoolingConverter : public GpuConverter { -public: - AdaptivePoolingConverter() {} - virtual ~AdaptivePoolingConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::adaptive_avg_pool1d(Tensor self, int[1] output_size) -> Tensor", - "aten::adaptive_avg_pool2d(Tensor self, int[2] output_size) -> Tensor", - "aten::adaptive_max_pool2d(Tensor self, int[2] output_size) -> (Tensor, Tensor)" - }; - } - - /** TODO: TRY TO SUPPORT SCHEMA PATTERNS BELLOW: - * aten::adaptive_avg_pool3d(Tensor self, int[3] output_size) -> Tensor - * aten::adaptive_max_pool1d(Tensor self, int[1] output_size) -> (Tensor, Tensor) - * aten::adaptive_max_pool3d(Tensor self, int[3] output_size) -> (Tensor, Tensor) - **/ - - const std::vector node_kind() { - return {torch::jit::aten::adaptive_avg_pool1d, - torch::jit::aten::adaptive_avg_pool2d, - torch::jit::aten::adaptive_max_pool2d - }; - } -}; - - - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/converter/gpu/reduce.cpp b/poros/poros/converter/gpu/reduce.cpp deleted file mode 100644 index 02b8f9214c5..00000000000 --- a/poros/poros/converter/gpu/reduce.cpp +++ /dev/null @@ -1,346 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file reduce.cpp -* @author tianjinjin@baidu.com -* @date Fri Aug 27 10:18:24 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/reduce.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/* -"aten::mean(Tensor self, *, ScalarType? dtype=None) -> Tensor", -"aten::mean.dim(Tensor self, int[1] dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor"*/ -bool MeanConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for MeanConverter is not Tensor as expected"); - - - //extract self - auto in_tensor = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in_tensor != nullptr), "Unable to init input tensor for node: " << *node); - auto in_dims = nvdim_to_sizes(in_tensor->getDimensions()); - LOG(WARNING) << "MeanConverter disregards dtype"; - - uint32_t axis_mask = (uint32_t)(((uint64_t)1 << in_dims.size()) - 1); - auto keepdim = false; - - // aten::mean.dim situation - auto maybe_dim = engine->context().get_constant(inputs[1]); - if ((inputs.size() == 4) && maybe_dim.isIntList() && - (engine->context().get_constant(inputs[2])).isBool()) { - auto dims = maybe_dim.toIntList(); - c10::List calculated_dims; - for (size_t i = 0; i < dims.size(); i++) { - auto dim_val = dims[i] < 0 ? (in_dims.size() + dims[i]) : dims[i]; - calculated_dims.push_back(dim_val); - } - axis_mask = 0; - for (size_t d = 0; d < calculated_dims.size(); d++) { - axis_mask |= 1 << calculated_dims[d]; - } - keepdim = (engine->context().get_constant(inputs[2])).toBool(); - } - - auto mean_layer = engine->network()->addReduce(*in_tensor, - nvinfer1::ReduceOperation::kAVG, axis_mask, keepdim); - POROS_CHECK(mean_layer, "Unable to create mean layer from node: " << *node); - mean_layer->setName((layer_info(node) + "_IReduceLayer_avg").c_str()); - - engine->context().set_tensor(node->outputs()[0], mean_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << mean_layer->getOutput(0)->getDimensions(); - return true; -} - - -/* -"aten::sum(Tensor self, *, ScalarType? dtype=None) -> Tensor", -"aten::sum.dim_IntList(Tensor self, int[1] dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor"*/ -bool SumConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for SumConverter is not Tensor as expected"); - - //extract self - auto in_tensor = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in_tensor != nullptr), "Unable to init input tensor for node: " << *node); - auto in_dims = nvdim_to_sizes(in_tensor->getDimensions()); - LOG(WARNING) << "SumConverter disregards dtype"; - - uint32_t axis_mask = (uint32_t)(((uint64_t)1 << in_dims.size()) - 1); - auto keepdim = false; - - // aten::sum.dim_IntList situation - auto maybe_dim = engine->context().get_constant(inputs[1]); - if ((inputs.size() == 4) && maybe_dim.isIntList() && - (engine->context().get_constant(inputs[2])).isBool()) { - auto dims = maybe_dim.toIntList(); - c10::List calculated_dims; - for (size_t i = 0; i < dims.size(); i++) { - auto dim_val = dims[i] < 0 ? (in_dims.size() + dims[i]) : dims[i]; - calculated_dims.push_back(dim_val); - } - axis_mask = 0; - for (size_t d = 0; d < calculated_dims.size(); d++) { - axis_mask |= 1 << calculated_dims[d]; - } - keepdim = (engine->context().get_constant(inputs[2])).toBool(); - } - - auto mean_layer = engine->network()->addReduce(*in_tensor, - nvinfer1::ReduceOperation::kSUM, axis_mask, keepdim); - POROS_CHECK(mean_layer, "Unable to create mean layer from node: " << *node); - mean_layer->setName((layer_info(node) + "_IReduceLayer_sum").c_str()); - - engine->context().set_tensor(node->outputs()[0], mean_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << mean_layer->getOutput(0)->getDimensions(); - return true; -} - -/* -"aten::prod(Tensor self, *, ScalarType? dtype=None) -> Tensor", -"aten::prod.dim_int(Tensor self, int dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor"*/ -bool ProdConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for ProdConverter is not Tensor as expected"); - - //extract self - auto in_tensor = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in_tensor != nullptr), "Unable to init input tensor for node: " << *node); - auto in_dims = nvdim_to_sizes(in_tensor->getDimensions()); - LOG(WARNING) << "ProdConverter disregards dtype"; - - uint32_t axis_mask = (uint32_t)(((uint64_t)1 << in_dims.size()) - 1); - auto keepdim = false; - - //aten::prod.dim_int situation - auto maybe_dim = engine->context().get_constant(inputs[1]); - if ((inputs.size() == 4) && maybe_dim.isInt() && - (engine->context().get_constant(inputs[2])).isBool()) { - auto dim = maybe_dim.toInt(); - dim = dim < 0 ? (in_tensor->getDimensions().nbDims + dim) : dim; - axis_mask = 1 << dim; - - keepdim = (engine->context().get_constant(inputs[2])).toBool(); - } - - auto mean_layer = engine->network()->addReduce(*in_tensor, - nvinfer1::ReduceOperation::kPROD, axis_mask, keepdim); - POROS_CHECK(mean_layer, "Unable to create mean layer from node: " << *node); - mean_layer->setName((layer_info(node) + "_IReduceLayer_prod").c_str()); - - engine->context().set_tensor(node->outputs()[0], mean_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << mean_layer->getOutput(0)->getDimensions(); - return true; -} - -/* -"aten::max(Tensor self) -> Tensor", -"aten::max.other(Tensor self, Tensor other) -> Tensor" -"aten::max.dim(Tensor self, int dim, bool keepdim=False) -> (Tensor values, Tensor indices)"*/ -bool MaxMinConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for MaxMinConverter is not Tensor as expected"); - - //extract self - auto in_tensor = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in_tensor != nullptr), "Unable to init input tensor for node: " << *node); - auto in_dims = nvdim_to_sizes(in_tensor->getDimensions()); - - bool is_dynamic = check_nvtensor_is_dynamic(in_tensor); - - nvinfer1::ILayer* new_layer; - //aten::max situation - if (inputs.size() == 1) { - uint32_t axis_mask = (uint32_t)(((uint64_t)1 << in_dims.size()) - 1); - auto keepdim = false; - - nvinfer1::ReduceOperation reduce_type = (node->kind() == torch::jit::aten::max) - ? nvinfer1::ReduceOperation::kMAX - : nvinfer1::ReduceOperation::kMIN; - new_layer = engine->network()->addReduce(*in_tensor, reduce_type, axis_mask, keepdim); - new_layer->setName((layer_info(node) + "_IReduceLayer_max_or_min").c_str()); - POROS_CHECK(new_layer, "Unable to create reduce layer from node: " << *node); - - //aten::max.other situation - } else if (inputs.size() == 2) { - //extract other - auto other = engine->context().get_tensor(inputs[1]); - POROS_CHECK_TRUE((other != nullptr), "Unable to init input tensor for node: " << *node); - - nvinfer1::ElementWiseOperation element_type = (node->kind() == torch::jit::aten::max) - ? nvinfer1::ElementWiseOperation::kMAX - : nvinfer1::ElementWiseOperation::kMIN; - new_layer = add_elementwise(engine, - element_type, - in_tensor, - other, - layer_info(node) + "_max_or_min"); - POROS_CHECK(new_layer, "Unable to create element_wise layer from node: " << *node); - - } else if (inputs.size() == 3 && node->outputs().size() == 2 && - inputs[1]->type()->kind() == c10::TypeKind::IntType) { - POROS_CHECK_TRUE((in_dims.size() > 1), - "Converter aten::max.dim error: At least 2 dimensions are required for input[0]."); - nvinfer1::ITensor* output_max = nullptr; - nvinfer1::ITensor* output_indices = nullptr; - int64_t dim = engine->context().get_constant(inputs[1]).toInt(); - dim = dim < 0 ? in_dims.size() + dim : dim; - - bool keep_dim = engine->context().get_constant(inputs[2]).toBool(); - uint32_t shiftDim = 1 << dim; - nvinfer1::TopKOperation topk_option = (node->kind() == torch::jit::aten::max) ? - nvinfer1::TopKOperation::kMAX : - nvinfer1::TopKOperation::kMIN; - nvinfer1::ITopKLayer* topk_layer = engine->network()->addTopK(*in_tensor, topk_option, 1, shiftDim); - POROS_CHECK(topk_layer, "Unable to create TopK layer from node: " << *node); - topk_layer->setName((layer_info(node) + "_ITopKLayer").c_str()); - output_max = topk_layer->getOutput(0); - output_indices = topk_layer->getOutput(1); - - // squeeze output dim - if (in_tensor->getDimensions().nbDims > 1 && !keep_dim) { - auto shuffle_layer1 = engine->network()->addShuffle(*output_max); - auto shuffle_layer2 = engine->network()->addShuffle(*output_indices); - if (is_dynamic) { - nvinfer1::ITensor* self_shape_tensor = engine->network()->addShape(*in_tensor)->getOutput(0); - nvinfer1::ITensor* squeeze_output_shape = squeeze_nv_shapetensor(engine, self_shape_tensor, dim); - shuffle_layer1->setInput(1, *squeeze_output_shape); - shuffle_layer2->setInput(1, *squeeze_output_shape); - } else { - in_dims.erase(in_dims.begin() + dim); - nvinfer1::Dims squeeze_output_dims = sizes_to_nvdim(in_dims); - shuffle_layer1->setReshapeDimensions(squeeze_output_dims); - shuffle_layer2->setReshapeDimensions(squeeze_output_dims); - } - output_max = shuffle_layer1->getOutput(0); - output_indices = shuffle_layer2->getOutput(0); - } - - engine->context().set_tensor(node->outputs()[0], output_max); - engine->context().set_tensor(node->outputs()[1], output_indices); - LOG(INFO) << "Output tensor1 shape: " << output_max->getDimensions(); - LOG(INFO) << "Output tensor2 shape: " << output_indices->getDimensions(); - return true; - - } else{ - //some other situation not supported yet - POROS_THROW_ERROR("We should never reach here for MaxMinConverter, meet Unsupported inputs size!"); - } - - nvinfer1::ITensor* output = new_layer->getOutput(0); - - if (output->getDimensions().nbDims == 0) { - auto shuffle_layer = engine->network()->addShuffle(*output); - nvinfer1::Dims output_dims; - output_dims.nbDims = 1; - output_dims.d[0] = 1; - shuffle_layer->setReshapeDimensions(output_dims); - shuffle_layer->setName((layer_info(node) + "_IShuffleLayer_for_output").c_str()); - output = shuffle_layer->getOutput(0); - } - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - return true; -} - -/* -"aten::argmax(Tensor self, int? dim=None, bool keepdim=False) -> (Tensor)"*/ -bool ArgmaxArgminConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for ArgmaxArgminConverter is not Tensor as expected"); - - // TODO: to imp dim=None - POROS_CHECK_TRUE((inputs[1]->type()->isSubtypeOf(c10::IntType::get())), - "input[1] for ArgmaxArgminConverter is not int as expected"); - - //extract self - auto in_tensor = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in_tensor != nullptr), "Unable to init input tensor for node: " << *node); - auto in_dims = nvdim_to_sizes(in_tensor->getDimensions()); - - bool is_dynamic = check_nvtensor_is_dynamic(in_tensor); - - POROS_CHECK_TRUE((in_dims.size() > 1), - "Converter aten::argmax error: At least 2 dimensions are required for input[0]."); - nvinfer1::ITensor* output_indices = nullptr; - - int64_t dim = 0; - dim = engine->context().get_constant(inputs[1]).toInt(); - dim = dim < 0 ? in_dims.size() + dim : dim; - bool keep_dim = engine->context().get_constant(inputs[2]).toBool(); - uint32_t shiftDim = 1 << dim; - - // nvinfer1::TopKOperation noly support kFLOAT, so this is transfer kINT32 to kFLOAT - if (in_tensor->getType() == nvinfer1::DataType::kINT32) { - auto id_layer = engine->network()->addIdentity(*in_tensor); - id_layer->setOutputType(0, nvinfer1::DataType::kFLOAT); - id_layer->setName((layer_info(node) + "_IIdentityLayer_int32_to_float").c_str()); - in_tensor = id_layer->getOutput(0); - } - - nvinfer1::TopKOperation topk_option = (node->kind() == torch::jit::aten::argmax) ? - nvinfer1::TopKOperation::kMAX : - nvinfer1::TopKOperation::kMIN; - nvinfer1::ITopKLayer* topk_layer = engine->network()->addTopK(*in_tensor, topk_option, 1, shiftDim); - POROS_CHECK(topk_layer, "Unable to create TopK layer from node: " << *node); - topk_layer->setName((layer_info(node) + "_ITopKLayer").c_str()); - output_indices = topk_layer->getOutput(1); - - // squeeze output dim - if (in_tensor->getDimensions().nbDims > 1 && !keep_dim) { - auto shuffle_layer = engine->network()->addShuffle(*output_indices); - if (is_dynamic) { - nvinfer1::ITensor* self_shape_tensor = engine->network()->addShape(*in_tensor)->getOutput(0); - nvinfer1::ITensor* squeeze_output_shape = squeeze_nv_shapetensor(engine, self_shape_tensor, dim); - shuffle_layer->setInput(1, *squeeze_output_shape); - } else { - in_dims.erase(in_dims.begin() + dim); - nvinfer1::Dims squeeze_output_dims = sizes_to_nvdim(in_dims); - shuffle_layer->setReshapeDimensions(squeeze_output_dims); - } - output_indices = shuffle_layer->getOutput(0); - } - engine->context().set_tensor(node->outputs()[0], output_indices); - LOG(INFO) << "Output tensor shape: " << output_indices->getDimensions(); - return true; -} - - -POROS_REGISTER_CONVERTER(TensorrtEngine, MeanConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, SumConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, ProdConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, MaxMinConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, ArgmaxArgminConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/reduce.h b/poros/poros/converter/gpu/reduce.h deleted file mode 100644 index 8bc1360d7a8..00000000000 --- a/poros/poros/converter/gpu/reduce.h +++ /dev/null @@ -1,160 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file reduce.h -* @author tianjinjin@baidu.com -* @date Fri Aug 27 10:18:24 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class MeanConverter : public GpuConverter { -public: - MeanConverter() {} - virtual ~MeanConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::mean(Tensor self, *, ScalarType? dtype=None) -> Tensor", - "aten::mean.dim(Tensor self, int[1] dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::mean.out(Tensor self, int[1] dim, bool keepdim=False, *, ScalarType? dtype=None, Tensor(a!) out) -> Tensor(a!)", - * "aten::mean.names_dim(Tensor self, Dimname[1] dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor", - * "aten::mean.names_out(Tensor self, Dimname[1] dim, bool keepdim=False, *, ScalarType? dtype=None, Tensor(a!) out) -> Tensor(a!)" - * **/ - const std::vector node_kind() { - return {torch::jit::aten::mean}; - } -}; - -class SumConverter : public GpuConverter { -public: - SumConverter() {} - virtual ~SumConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::sum(Tensor self, *, ScalarType? dtype=None) -> Tensor", - "aten::sum.dim_IntList(Tensor self, int[1] dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::sum.dim_DimnameList(Tensor self, Dimname[1] dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor", - * "aten::sum.IntList_out(Tensor self, int[1] dim, bool keepdim=False, *, ScalarType? dtype=None, Tensor(a!) out) -> Tensor(a!)", - * "aten::sum.DimnameList_out(Tensor self, Dimname[1] dim, bool keepdim=False, *, ScalarType? dtype=None, Tensor(a!) out) -> Tensor(a!)" - * **/ - const std::vector node_kind() { - return {torch::jit::aten::sum}; - } -}; - -class ProdConverter : public GpuConverter { -public: - ProdConverter() {} - virtual ~ProdConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::prod(Tensor self, *, ScalarType? dtype=None) -> Tensor", - "aten::prod.dim_int(Tensor self, int dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::prod.int_out(Tensor self, int dim, bool keepdim=False, *, ScalarType? dtype=None, Tensor(a!) out) -> Tensor(a!)", - * "aten::prod.dim_Dimname(Tensor self, Dimname dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor", - * "aten::prod.Dimname_out(Tensor self, Dimname dim, bool keepdim=False, *, ScalarType? dtype=None, Tensor(a!) out) -> Tensor(a!)" - * **/ - const std::vector node_kind() { - return {torch::jit::aten::prod}; - } -}; - -class MaxMinConverter : public GpuConverter { -public: - MaxMinConverter() {} - virtual ~MaxMinConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::max(Tensor self) -> Tensor", - "aten::min(Tensor self) -> Tensor", - "aten::max.other(Tensor self, Tensor other) -> Tensor", - "aten::min.other(Tensor self, Tensor other) -> Tensor", - "aten::max.dim(Tensor self, int dim, bool keepdim=False) -> (Tensor values, Tensor indices)", - "aten::min.dim(Tensor self, int dim, bool keepdim=False) -> (Tensor values, Tensor indices)",}; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::max.dim_max(Tensor self, int dim, bool keepdim=False, *, Tensor(a!) max, Tensor(b!) max_values) -> (Tensor(a!) values, Tensor(b!) indices)", - * "aten::max.names_dim(Tensor self, Dimname dim, bool keepdim=False) -> (Tensor values, Tensor indices)", - * "aten::max.names_dim_max(Tensor self, Dimname dim, bool keepdim=False, *, Tensor(a!) max, Tensor(b!) max_values) -> (Tensor(a!) values, Tensor(b!) indices)", - * "aten::max.out(Tensor self, Tensor other, *, Tensor(a!) out) -> Tensor(a!)" - * - * "aten::min.dim_min(Tensor self, int dim, bool keepdim=False, *, Tensor(a!) min, Tensor(b!) min_indices) -> (Tensor(a!) values, Tensor(b!) indices)", - * "aten::min.names_dim(Tensor self, Dimname dim, bool keepdim=False) -> (Tensor values, Tensor indices)", - * "aten::min.names_dim_min(Tensor self, Dimname dim, bool keepdim=False, *, Tensor(a!) min, Tensor(b!) min_indices) -> (Tensor(a!) values, Tensor(b!) indices)", - * "aten::min.out(Tensor self, Tensor other, *, Tensor(a!) out) -> Tensor(a!)", - * **/ - const std::vector node_kind() { - return {torch::jit::aten::max, - torch::jit::aten::min}; - } -}; - -class ArgmaxArgminConverter : public GpuConverter { -public: - ArgmaxArgminConverter() {} - virtual ~ArgmaxArgminConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::argmax(Tensor self, int dim, bool keepdim=False) -> (Tensor)", - "aten::argmax(Tensor self, int? dim=None, bool keepdim=False) -> (Tensor)", - "aten::argmin(Tensor self, int dim, bool keepdim=False) -> (Tensor)", - }; - } - - const std::vector node_kind() { - return {torch::jit::aten::argmax, - torch::jit::aten::argmin}; - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/reflection_pad.cpp b/poros/poros/converter/gpu/reflection_pad.cpp deleted file mode 100644 index a2965dfcac7..00000000000 --- a/poros/poros/converter/gpu/reflection_pad.cpp +++ /dev/null @@ -1,436 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file reflection_pad.cpp -* @author tianshaoqing@baidu.com -* @date Tue Aug 16 16:54:20 CST 2022 -* @brief -**/ - -#include "poros/converter/gpu/reflection_pad.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/** - * @brief 翻转input的dim维,支持dynamic - * - * @param [in] engine : trtengine - * @param [in] node : 当前节点 - * @param [in] input : 要翻转的tensor - * @param [in] is_dynamic : 输入是否是dynamic的 - * @param [in] dim : 指定要翻转的维度 - * - * @return nvinfer1::ITensor* - * @retval 返回翻转后的tensor -**/ -static nvinfer1::ITensor* flip_nvtensor(TensorrtEngine* engine, - const torch::jit::Node *node, - nvinfer1::ITensor* input, - bool is_dynamic, - int dim) { - - auto in_dims = input->getDimensions(); - int64_t in_rank = in_dims.nbDims; - dim = dim < 0 ? in_rank + dim : dim; - - POROS_ASSERT(dim >= 0 && dim < in_rank, "flip dim is out of range. expect range is [" + - std::to_string(-in_rank) + ", " + std::to_string(in_rank - 1) + "]."); - - if (!is_dynamic) { - std::vector start_vec, size_vec, stride_vec; - for (int32_t r = 0; r < in_rank; r++) { - start_vec.push_back(0); - size_vec.push_back(in_dims.d[r]); - stride_vec.push_back(1); - } - start_vec[dim] = size_vec[dim] - 1; - stride_vec[dim] = -1; - - auto slice_layer = engine->network()->addSlice(*input, - sizes_to_nvdim(start_vec), - sizes_to_nvdim(size_vec), - sizes_to_nvdim(stride_vec)); - slice_layer->setName((layer_info(node) + "_ISliceLayer_flip_for_dim_" + std::to_string(dim)).c_str()); - return slice_layer->getOutput(0); - } else { - nvinfer1::ITensor* input_shape = engine->network()->addShape(*input)->getOutput(0); - std::vector stride_vec(in_rank, 1), dim_mask_vec(in_rank, 0), tmp_vec(in_rank, 0); - stride_vec[dim] = -1; - dim_mask_vec[dim] = 1; - - nvinfer1::ITensor* dim_mask_tensor = tensor_to_const(engine, torch::tensor(dim_mask_vec, torch::kInt32)); - nvinfer1::ITensor* stride_tensor = tensor_to_const(engine, torch::tensor(stride_vec, torch::kInt32)); - - nvinfer1::ITensor* start_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - input_shape, - dim_mask_tensor, - layer_info(node) + "_prod_flip_for_dim_" + - std::to_string(dim))->getOutput(0); - - start_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUB, - start_tensor, - dim_mask_tensor, - layer_info(node) + "_sub_flip_for_dim_" + - std::to_string(dim))->getOutput(0); - - auto slice_layer = engine->network()->addSlice(*input, - sizes_to_nvdim(tmp_vec), - sizes_to_nvdim(tmp_vec), - sizes_to_nvdim(tmp_vec)); - slice_layer->setInput(0, *input); - slice_layer->setInput(1, *start_tensor); - slice_layer->setInput(2, *input_shape); - slice_layer->setInput(3, *stride_tensor); - slice_layer->setName((layer_info(node) + "_ISliceLayer_flip_for_dim_" + std::to_string(dim)).c_str()); - - return slice_layer->getOutput(0); - } -} - -/** - * @brief 根据padding的大小,计算left slice 的 start 和 size,以及 righit slice 的 size, - * 具体作用见ReflectionPadConverter::converter注释 - * - * @param [in] engine : trtengine - * @param [in] node : 当前节点 - * @param [in] input : 要 reflection padding 的 tensor - * @param [in] padding_is_nvtensor : padding参数是否是以nvscalar形式输入的 - * @param [in] padding_tensor : padding_is_nvtensor 为 true 时,读取内部 padding 数据 - * @param [in] padding_size : padding_is_nvtensor 为 false 时,读取内部 padding 数据 - * @param [in] axis : 当前 padding 的维度 - * - * @return std::tuple - * @retval 返回left slice 的 start 和 size,以及 righit slice 的 size -**/ -static std::tuple gen_slice_start_size(TensorrtEngine* engine, - const torch::jit::Node *node, - nvinfer1::ITensor* input, - bool padding_is_nvtensor, - std::vector padding_tensor, - std::vector padding_size, - int32_t axis) { - nvinfer1::ITensor* left_start_tensor = nullptr; - nvinfer1::ITensor* left_size_tensor = nullptr; - nvinfer1::ITensor* right_size_tensor = nullptr; - // start_vec[axis] = inDims.d[axis] - padding[padding_index] - 1; 基础是0 - // size_vec[axis] = padding[padding_index]; 基础是size - nvinfer1::ITensor* input_shape = engine->network()->addShape(*input)->getOutput(0); - - auto in_dims = input->getDimensions(); - int64_t in_rank = in_dims.nbDims; - std::vector dim_mask_vec(in_rank, 0), dim_remask_vec(in_rank, 1); - dim_mask_vec[axis] = 1; - dim_remask_vec[axis] = 0; - nvinfer1::ITensor* dim_mask_tensor = tensor_to_const(engine, torch::tensor(dim_mask_vec, torch::kInt32)); - nvinfer1::ITensor* dim_remask_tensor = tensor_to_const(engine, torch::tensor(dim_remask_vec, torch::kInt32)); - - nvinfer1::ITensor* shape_mask_axis = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - input_shape, - dim_mask_tensor, - layer_info(node) + std::string("_left_shape_mask_axis_") + - std::to_string(axis))->getOutput(0); - - nvinfer1::ITensor* left_padding_size_tensor = nullptr; - nvinfer1::ITensor* right_padding_size_tensor = nullptr; - if (!padding_is_nvtensor) { - left_padding_size_tensor = tensor_to_const(engine, torch::tensor({padding_size[0]}, torch::kInt32)); - right_padding_size_tensor = tensor_to_const(engine, torch::tensor({padding_size[1]}, torch::kInt32)); - } else { - left_padding_size_tensor = padding_tensor[0]; - right_padding_size_tensor = padding_tensor[1]; - } - - nvinfer1::ITensor* left_padding_mask_axis = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - dim_mask_tensor, - left_padding_size_tensor, - layer_info(node) + std::string("_left_padding_mask_axis_") + - std::to_string(axis))->getOutput(0); - - nvinfer1::ITensor* shape_sub_left_padding_mask_axis = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUB, - shape_mask_axis, - left_padding_mask_axis, - layer_info(node) + std::string("_left_shape_sub_padding_mask_axis_") + - std::to_string(axis))->getOutput(0); - - left_start_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUB, - shape_sub_left_padding_mask_axis, - dim_mask_tensor, - layer_info(node) + std::string("_left_shape_sub_padding_sub_one_mask_axis_") + - std::to_string(axis))->getOutput(0); - - nvinfer1::ITensor* shape_remask_axis = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - input_shape, - dim_remask_tensor, - layer_info(node) + std::string("_left_shape_remask_axis_") + - std::to_string(axis))->getOutput(0); - - left_size_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - shape_remask_axis, - left_padding_mask_axis, - layer_info(node) + std::string("_left_shape_remask_sum_padding_axis_") + - std::to_string(axis))->getOutput(0); - - nvinfer1::ITensor* right_padding_mask_axis = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - dim_mask_tensor, - right_padding_size_tensor, - layer_info(node) + std::string("_right_padding_mask_axis_") + - std::to_string(axis))->getOutput(0); - - right_size_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - shape_remask_axis, - right_padding_mask_axis, - layer_info(node) + std::string("_right_shape_remask_sum_padding_axis_") + - std::to_string(axis))->getOutput(0); - - return std::make_tuple(left_start_tensor, left_size_tensor, right_size_tensor); -} - -/** - * @brief ReflectionPad功能:镜像填充,pad规则类似constant_pad_nd,只不过pad值换成边缘的镜像 - * 例如:输入 x = torch.arange(8).reshape(2, 4) = - * [[0,1,2,3], - * [4,5,6,7]] - * 那么 ReflectionPad1d(x, [1,2]) = - * [[1,0,1,2,3,2,1], - * [5,4,5,6,7,6,5]] - * - * converter实现思路 - * 先将x整体按照pad维度翻转x_flip = - * [[3,2,1,0], - * [7,6,5,4]] - * 左边padding size = 1 - * x' = cat([x_flip[:, -2], x], dim = 1) - * [[3,2,|1|,0], [[|1|,0,1,2,3], - * [7,6,|5|,4]] [|5|,4,5,6,7]] - * ^ ^ - * 右边padding size = 2 - * x'' = cat([x', x_flip[:, 1:2]], dim = 1) - * [[3,|2,1|,0], [[1,0,1,2,3,|2,1|], - * [7,|6,5|,4]] [5,4,5,6,7,|6,5|]] - * ^ ^ - * ReflectionPad2d同理 - * - * @param [in] engine : trtengine - * @param [in] node : 当前节点 - * - * @return bool - * @retval true convert 成功,false convert 失败 -**/ -bool ReflectionPadConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - // "aten::reflection_pad1d(Tensor self, int[2] padding) -> Tensor" - // "aten::replication_pad2d(Tensor self, int[4] padding) -> Tensor" - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for ReflectionPadConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for ReflectionPadConverter is not Tensor as expected"); - - //extract self - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - auto inDims = in->getDimensions(); - int64_t inRank = inDims.nbDims; - - std::vector tensors_vec; - - bool has_tensor_scalar = check_inputs_tensor_scalar(engine, node); - bool input0_is_dynamic = check_nvtensor_is_dynamic(in); - - if (!has_tensor_scalar) { - - //extract padding - auto padding = (engine->context().get_constant(inputs[1])).toIntList().vec(); - - for (int64_t i = 0; i < int(padding.size() / 2); i++) { - int64_t axis = inRank - (i + 1); // axis = {inRank - 1, inRank - 2} - int64_t padding_index = i * 2; - - nvinfer1::ITensor* in_flip = flip_nvtensor(engine, node, in, input0_is_dynamic, axis); - - std::tuple left_start_size_right_size; - - if (inDims.d[axis] < 0) { - std::vector tmp_itensor_vec; - std::vector padding_size_vec = {(int32_t)padding[padding_index], (int32_t)padding[padding_index + 1]}; - left_start_size_right_size = gen_slice_start_size(engine, node, in, false, tmp_itensor_vec, padding_size_vec, axis); - } - - if (padding[padding_index] > 0) { // left padding value - tensors_vec.clear(); - std::vector start_vec, size_vec, stride_vec; - for (int32_t r = 0; r < inRank; r++) { - start_vec.push_back(0); - size_vec.push_back(inDims.d[r]); - stride_vec.push_back(1); - } - start_vec[axis] = inDims.d[axis] - padding[padding_index] - 1; - size_vec[axis] = padding[padding_index]; - - auto slice_layer = engine->network()->addSlice(*in_flip, - sizes_to_nvdim(start_vec), - sizes_to_nvdim(size_vec), - sizes_to_nvdim(stride_vec)); - slice_layer->setName((layer_info(node) + "_ISliceLayer_for_leftpadding_" + std::to_string(axis)).c_str()); - if (inDims.d[axis] < 0) { - slice_layer->setInput(1, *(std::get<0>(left_start_size_right_size))); - slice_layer->setInput(2, *(std::get<1>(left_start_size_right_size))); - } - - tensors_vec.push_back(slice_layer->getOutput(0)); - tensors_vec.push_back(in); - - auto concat_layer = engine->network()->addConcatenation(tensors_vec.data(), tensors_vec.size()); - concat_layer->setAxis(axis); - concat_layer->setName((layer_info(node) + "_IConcatenationLayer_for_leftpadding_" + std::to_string(axis)).c_str()); - in = concat_layer->getOutput(0); - inDims = in->getDimensions(); - } - - if (padding[padding_index + 1] > 0) { // right padding value - tensors_vec.clear(); - tensors_vec.push_back(in); - - std::vector start_vec, size_vec, stride_vec; - for (int32_t r = 0; r < inRank; r++) { - start_vec.push_back(0); - size_vec.push_back(inDims.d[r]); - stride_vec.push_back(1); - } - start_vec[axis] = 1; - size_vec[axis] = padding[padding_index + 1]; - - auto slice_layer = engine->network()->addSlice(*in_flip, - sizes_to_nvdim(start_vec), - sizes_to_nvdim(size_vec), - sizes_to_nvdim(stride_vec)); - slice_layer->setName((layer_info(node) + "_ISliceLayer_for_rightpadding_"+ std::to_string(axis)).c_str()); - if (inDims.d[axis] < 0) { - slice_layer->setInput(2, *(std::get<2>(left_start_size_right_size))); - } - - tensors_vec.push_back(slice_layer->getOutput(0)); - - auto concat_layer = engine->network()->addConcatenation(tensors_vec.data(), tensors_vec.size()); - concat_layer->setAxis(axis); - concat_layer->setName((layer_info(node) + "_IConcatenationLayer_for_rightpadding_" + std::to_string(axis)).c_str()); - in = concat_layer->getOutput(0); - inDims = in->getDimensions(); - - } - } - } else { - // 先分开 - std::vector padding_tensor_vec; - nvinfer1::ITensor* padding_tensor = this->get_tensor_scalar(inputs[1]); - nvinfer1::Dims padding_tensor_dims = padding_tensor->getDimensions(); - - for (int i = 0; i < padding_tensor_dims.d[0]; i++) { - std::vector start_vec(1, i), size_vec(1, 1), stride_vec(1, 1); - auto slice_layer = engine->network()->addSlice(*padding_tensor, - sizes_to_nvdim(start_vec), - sizes_to_nvdim(size_vec), - sizes_to_nvdim(stride_vec)); - slice_layer->setName((layer_info(node) + "_ISliceLayer_for_padding_tensor_" + std::to_string(i)).c_str()); - padding_tensor_vec.push_back(slice_layer->getOutput(0)); - } - - for (size_t i = 0; i < padding_tensor_vec.size() / 2; i++) { - int64_t axis = inRank - (i + 1); // axis = {inRank - 1, inRank - 2} - int64_t padding_index = i * 2; - - nvinfer1::ITensor* in_flip = flip_nvtensor(engine, node, in, input0_is_dynamic, axis); - - std::vector itensor_vec = {padding_tensor_vec[padding_index], padding_tensor_vec[padding_index + 1]}; - std::vector tmp_vec; - auto left_start_size_right_size = gen_slice_start_size(engine, node, in, true, itensor_vec, tmp_vec, axis); - - // left - tensors_vec.clear(); - std::vector start_vec, size_vec, stride_vec(inRank, 1); - - auto slice_layer = engine->network()->addSlice(*in_flip, - sizes_to_nvdim(stride_vec), - sizes_to_nvdim(stride_vec), - sizes_to_nvdim(stride_vec)); - - slice_layer->setInput(1, *(std::get<0>(left_start_size_right_size))); - slice_layer->setInput(2, *(std::get<1>(left_start_size_right_size))); - slice_layer->setName((layer_info(node) + "_ISliceLayer_for_leftpadding_" + std::to_string(axis)).c_str()); - - tensors_vec.push_back(slice_layer->getOutput(0)); - tensors_vec.push_back(in); - - auto concat_layer = engine->network()->addConcatenation(tensors_vec.data(), tensors_vec.size()); - concat_layer->setAxis(axis); - in = concat_layer->getOutput(0); - concat_layer->setName((layer_info(node) + "_IConcatenationLayer_for_leftpadding_" + std::to_string(axis)).c_str()); - inDims = in->getDimensions(); - - // right - tensors_vec.clear(); - tensors_vec.push_back(in); - - std::vector start_vec2, stride_vec2; - for (int32_t r = 0; r < inRank; r++) { - start_vec2.push_back(0); - stride_vec2.push_back(1); - } - start_vec2[axis] = 1; - - auto slice_layer2 = engine->network()->addSlice(*in_flip, - sizes_to_nvdim(start_vec2), - sizes_to_nvdim(stride_vec2), - sizes_to_nvdim(stride_vec2)); - - slice_layer2->setInput(2, *(std::get<2>(left_start_size_right_size))); - slice_layer2->setName((layer_info(node) + "_ISliceLayer_for_rightpadding_" + std::to_string(axis)).c_str()); - tensors_vec.push_back(slice_layer2->getOutput(0)); - - auto concat_layer2 = engine->network()->addConcatenation(tensors_vec.data(), tensors_vec.size()); - concat_layer2->setAxis(axis); - concat_layer2->setName((layer_info(node) + "_IConcatenationLayer_for_rightpadding_" + std::to_string(axis)).c_str()); - in = concat_layer2->getOutput(0); - inDims = in->getDimensions(); - } - } - - engine->context().set_tensor(node->outputs()[0], in); - LOG(INFO) << "Output tensor shape: " << in->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, ReflectionPadConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/reflection_pad.h b/poros/poros/converter/gpu/reflection_pad.h deleted file mode 100644 index d49133025f0..00000000000 --- a/poros/poros/converter/gpu/reflection_pad.h +++ /dev/null @@ -1,65 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file reflection_pad.h -* @author tianshaoqing@baidu.com -* @date Tue Aug 16 16:54:20 CST 2022 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class ReflectionPadConverter : public GpuConverter { -public: - ReflectionPadConverter() {} - virtual ~ReflectionPadConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::reflection_pad1d(Tensor self, int[2] padding) -> Tensor", - "aten::reflection_pad2d(Tensor self, int[4] padding) -> Tensor", - }; - } - - const std::vector node_kind() { - return {torch::jit::aten::reflection_pad1d, - torch::jit::aten::reflection_pad2d, - }; - } - - bool assign_schema_attr() { - bool result = true; - result &= assign_schema_attr_helper({{"aten::reflection_pad1d(Tensor self, int[2] padding) -> Tensor", {1, 1}}}); - result &= assign_schema_attr_helper({{"aten::reflection_pad2d(Tensor self, int[4] padding) -> Tensor", {1, 1}}}); - return result; - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/replication_pad.cpp b/poros/poros/converter/gpu/replication_pad.cpp deleted file mode 100644 index f3ec4e2c60c..00000000000 --- a/poros/poros/converter/gpu/replication_pad.cpp +++ /dev/null @@ -1,144 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -// Part of the following code in this file refs to -// https://github.com/pytorch/TensorRT/blob/master/core/conversion/converters/impl/replication_pad.cpp -// -// Copyright (c) 2020-present, NVIDIA CORPORATION. All rights reserved. -// Copyright (c) Meta Platforms, Inc. and affiliates. -// Licensed under the 3-Clause BSD License - -/** -* @file replication_pad.cpp -* @author tianjinjin@baidu.com -* @date Tue Sep 7 14:29:20 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/replication_pad.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/* -"aten::replication_pad1d(Tensor self, int[2] padding) -> Tensor", -"aten::replication_pad2d(Tensor self, int[4] padding) -> Tensor", -"aten::replication_pad3d(Tensor self, int[6] padding) -> Tensor", -*/ -bool ReplicationPadConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for ReplicationPadConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for ReplicationPadConverter is not Tensor as expected"); - - //extract self - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - auto inDims = in->getDimensions(); - int64_t inRank = inDims.nbDims; - - //extract padding - auto padding = (engine->context().get_constant(inputs[1])).toIntList().vec(); - if (padding.size() == 1) { - POROS_THROW_ERROR("Only 3D, 4D, 5D padding with non-constant padding are supported for now"); - } - if (inRank == 3) { - POROS_CHECK(padding.size() == 2, "3D tensors expect 2 values for padding"); - } else if (inRank == 4) { - POROS_CHECK(padding.size() == 4, "4D tensors expect 4 values for padding"); - } else if (inRank == 5) { - POROS_CHECK(padding.size() == 6, "5D tensors expect 6 values for padding"); - } else { - POROS_THROW_ERROR("Only 3D, 4D, 5D padding with non-constant padding are supported for now"); - } - - std::vector tensors_vec; - // input: (N, C, D_in, H_in, W_in). - // padding: (padding_left, padding_right, padding_top, padding_bottom, padding_front, padding_back) - // When axis is inRank - 1, making W_out = W_in + padding_left + padding_right. - // When axis is inRank - 2, making H_out = H_in + padding_top + padding_bottom. - // When axis is inRank - 1, making D_out = D_in + padding_front + padding_back. - for (int64_t i = 0; i < int(padding.size() / 2); i++) { - int64_t axis = inRank - (i + 1); // axis = {inRank - 1, inRank - 2, inRank - 3} - int64_t padding_index = i * 2; - - if (padding[padding_index] > 0) { // left/top/front padding value - tensors_vec.clear(); - at::Tensor left_indices = torch::tensor({0}, torch::kInt32); - auto indicesTensor = tensor_to_const(engine, left_indices); - auto left_gather_layer = engine->network()->addGather(*in, *indicesTensor, axis); - left_gather_layer->setName((layer_info(node) + "_IGatherLayer_for_left_axis_" + std::to_string(axis)).c_str()); - auto left_gather_out = left_gather_layer->getOutput(0); - for (int i = 0; i < padding[padding_index]; i++) { - tensors_vec.push_back(left_gather_out); - } - tensors_vec.push_back(in); - auto concat_layer = engine->network()->addConcatenation(tensors_vec.data(), tensors_vec.size()); - concat_layer->setAxis(axis); - concat_layer->setName((layer_info(node) + "_IConcatenationLayer_for_left_axis_" + std::to_string(axis)).c_str()); - in = concat_layer->getOutput(0); - inDims = in->getDimensions(); - } - - if (padding[padding_index + 1] > 0) { // right/bottom/back padding value - tensors_vec.clear(); - tensors_vec.push_back(in); - - nvinfer1::ITensor* indicesTensor = NULL; - if (inDims.d[axis] == -1) { - auto shapeTensor = engine->network()->addShape(*in)->getOutput(0); - at::Tensor dimValue = torch::tensor({axis}, torch::kInt32); - auto dimTensor = tensor_to_const(engine, dimValue); - indicesTensor = engine->network()->addGather(*shapeTensor, *dimTensor, 0)->getOutput(0); - auto oneTensor = tensor_to_const(engine, torch::tensor({1}, torch::kInt32)); - indicesTensor = engine->network()->addElementWise(*indicesTensor, - *oneTensor, nvinfer1::ElementWiseOperation::kSUB)->getOutput(0); - } else { - auto indices = torch::tensor({inDims.d[axis] - 1}, torch::kInt32); - indicesTensor = tensor_to_const(engine, indices); - } - auto right_gather_layer = engine->network()->addGather(*in, *indicesTensor, axis); - right_gather_layer->setName((layer_info(node) + "_IGatherLayer_for_right_axis_" + std::to_string(axis)).c_str()); - auto right_gather_out = right_gather_layer->getOutput(0); - - for (int i = 0; i < padding[padding_index + 1]; i++) { - tensors_vec.push_back(right_gather_out); - } - - auto concat_layer = engine->network()->addConcatenation(tensors_vec.data(), tensors_vec.size()); - concat_layer->setAxis(axis); - concat_layer->setName((layer_info(node) + "_IConcatenationLayer_for_right_axis_" + std::to_string(axis)).c_str()); - in = concat_layer->getOutput(0); - inDims = in->getDimensions(); - } - } - - engine->context().set_tensor(node->outputs()[0], in); - LOG(INFO) << "Output tensor shape: " << in->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, ReplicationPadConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/replication_pad.h b/poros/poros/converter/gpu/replication_pad.h deleted file mode 100644 index 1f9fe8bc43f..00000000000 --- a/poros/poros/converter/gpu/replication_pad.h +++ /dev/null @@ -1,65 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file replication_pad.h -* @author tianjinjin@baidu.com -* @date Tue Sep 7 14:29:20 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class ReplicationPadConverter : public GpuConverter { -public: - ReplicationPadConverter() {} - virtual ~ReplicationPadConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::replication_pad1d(Tensor self, int[2] padding) -> Tensor", - "aten::replication_pad2d(Tensor self, int[4] padding) -> Tensor", - "aten::replication_pad3d(Tensor self, int[6] padding) -> Tensor", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::replication_pad1d.out(Tensor self, int[2] padding, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::replication_pad2d.out(Tensor self, int[4] padding, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::replication_pad3d.out(Tensor self, int[6] padding, *, Tensor(a!) out) -> Tensor(a!)" - * **/ - const std::vector node_kind() { - return {torch::jit::aten::replication_pad1d, - torch::jit::aten::replication_pad2d, - torch::jit::aten::replication_pad3d, - }; - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/roll.cpp b/poros/poros/converter/gpu/roll.cpp deleted file mode 100644 index 7340c15ed43..00000000000 --- a/poros/poros/converter/gpu/roll.cpp +++ /dev/null @@ -1,114 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file roll.cpp -* @author tianshaoqing@baidu.com -* @date Wed Jul 20 16:34:51 CST 2022 -* @brief -**/ - -#include "poros/converter/gpu/roll.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -// aten::roll(Tensor self, int[1] shifts, int[1] dims=[]) -> (Tensor) -bool RollConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 3), "invaid inputs size for RollConverter"); - - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for RollConverter is not Tensor as expected"); - - POROS_CHECK_TRUE((inputs[1]->type()->isSubtypeOf(c10::ListType::ofInts()) - && inputs[2]->type()->isSubtypeOf(c10::ListType::ofInts())), - "input[1] or input[2] for RollConverter is not int[] as expected"); - // extract self - nvinfer1::ITensor* self = engine->context().get_tensor(inputs[0]); - // extract shifts - std::vector shifts_vec = (engine->context().get_constant(inputs[1])).toIntList().vec(); - // extract dims - std::vector dims_vec = (engine->context().get_constant(inputs[2])).toIntList().vec(); - - POROS_CHECK_TRUE((shifts_vec.size() == dims_vec.size()), - "The length of shifts and dims must be equal in RollConverter."); - - // Implementation of aten::roll - // example: - // input = {1, 2, 3, 4, 5}; shifts = 3; dim = 0; - // Then slice input into two parts: {1, 2} and {3, 4, 5}. - // Finally flip their order and concat them on rolling dim 0: {3, 4, 5, 1, 2}. - // And so on when multiple dimensions. - nvinfer1::Dims self_dims = self->getDimensions(); - for (size_t i = 0; i < shifts_vec.size(); i++) { - std::vector tensorlist; - int64_t rolling_dim = dims_vec[i]; - rolling_dim = (rolling_dim < 0) ? (self_dims.nbDims + rolling_dim) : rolling_dim; - - int64_t shift_stride = shifts_vec[i]; - // Shift is allowed to be greater than the rolling dimension, so we need to take the remainder. - shift_stride = shift_stride % self_dims.d[rolling_dim]; - // when shift == 0, on processing required - if (shift_stride == 0) { - continue; - } - std::vector start_vec(self_dims.nbDims, 0); - std::vector size_vec(self_dims.nbDims, 0); - std::vector stride_vec(self_dims.nbDims, 1); - - for (int32_t s = 0; s < self_dims.nbDims; s++) { - size_vec[s] = self_dims.d[s]; - } - - size_vec[rolling_dim] = (shift_stride < 0) ? (-shift_stride) : (self_dims.d[rolling_dim] - shift_stride); - - auto slice_left_layer = engine->network()->addSlice(*self, - sizes_to_nvdim(start_vec), - sizes_to_nvdim(size_vec), - sizes_to_nvdim(stride_vec)); - slice_left_layer->setName((layer_info(node) + "_left_slice_" + std::to_string(i)).c_str()); - nvinfer1::ITensor* left_slice = slice_left_layer->getOutput(0); - - start_vec[rolling_dim] = size_vec[rolling_dim]; - size_vec[rolling_dim] = self_dims.d[rolling_dim] - size_vec[rolling_dim]; - - auto slice_right_layer = engine->network()->addSlice(*self, - sizes_to_nvdim(start_vec), - sizes_to_nvdim(size_vec), - sizes_to_nvdim(stride_vec)); - slice_right_layer->setName((layer_info(node) + "_right_slice_" + std::to_string(i)).c_str()); - nvinfer1::ITensor* right_slice = slice_right_layer->getOutput(0); - tensorlist.push_back(right_slice); - tensorlist.push_back(left_slice); - - auto cat_layer = engine->network()->addConcatenation(tensorlist.data(), tensorlist.size()); - cat_layer->setAxis(static_cast(rolling_dim)); - cat_layer->setName((layer_info(node) + "_cat_" + std::to_string(i)).c_str()); - self = cat_layer->getOutput(0); - } - - engine->context().set_tensor(node->outputs()[0], self); - LOG(INFO) << "Output shape: " << self->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, RollConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/roll.h b/poros/poros/converter/gpu/roll.h deleted file mode 100644 index a55da88eb9c..00000000000 --- a/poros/poros/converter/gpu/roll.h +++ /dev/null @@ -1,57 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file roll.h -* @author tianshaoqing@baidu.com -* @date Wed Jul 20 16:33:51 CST 2022 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class RollConverter : public GpuConverter { -public: - RollConverter() {} - virtual ~RollConverter() {} - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - const std::vector schema_string() { - return {"aten::roll(Tensor self, int[1] shifts, int[1] dims=[]) -> Tensor"}; - } - - const std::vector node_kind() { - // return {torch::jit::aten::roll}; // can't find defintion in torch-1.9.0 - return {c10::Symbol::fromQualString("aten::roll")}; - } - - bool assign_schema_attr() { - return assign_schema_attr_helper({{"aten::roll(Tensor self, int[1] shifts, int[1] dims=[]) -> Tensor", {0, 0}}}); - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/select.cpp b/poros/poros/converter/gpu/select.cpp deleted file mode 100644 index e9d5778fc41..00000000000 --- a/poros/poros/converter/gpu/select.cpp +++ /dev/null @@ -1,1318 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file select.cpp -* @author tianjinjin@baidu.com -* @date Tue Aug 24 16:31:28 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/select.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/*aten::select.int(Tensor(a) self, int dim, int index) -> Tensor(a)*/ -bool SelectConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 3), "invaid inputs size for SelectConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for SelectConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[1]->node()->kind() == torch::jit::prim::Constant), - "input[1] for SelectConverter is not come from prim::Constant as expected"); - // POROS_CHECK_TRUE((inputs[2]->node()->kind() == torch::jit::prim::Constant), - // "input[2] for SelectConverter is not come from prim::Constant as expected"); - - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - auto maxDim = static_cast(in->getDimensions().nbDims); - - //extract dim - auto dim = (engine->context().get_constant(inputs[1])).toInt(); - dim = dim < 0 ? dim + maxDim : dim; - - nvinfer1::ITensor* index_tensor = engine->context().get_tensor(inputs[2]); - //extract index - if (index_tensor == nullptr) { - auto ind = (int32_t)((engine->context().get_constant(inputs[2])).toInt()); - // dynamic情况下 dim这一维是动态的-1,且index为倒序,需要转正 - if (in->getDimensions().d[dim] < 0 && ind < 0) { - nvinfer1::ITensor* in_shape_tensor = engine->network()->addShape(*in)->getOutput(0); - std::vector start_vec = {dim}, size_vec = {1}, stride_vec = {1}; - nvinfer1::ISliceLayer* slice_layer = engine->network()->addSlice(*in_shape_tensor, - sizes_to_nvdim(start_vec), - sizes_to_nvdim(size_vec), - sizes_to_nvdim(stride_vec)); - nvinfer1::ITensor* in_dim_val = slice_layer->getOutput(0); - nvinfer1::ITensor* ind_tensor = tensor_to_const(engine, torch::tensor({ind}).to(torch::kI32)); - index_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - in_dim_val, - ind_tensor, - layer_info(node) + std::string("_neg_index_to_pos"))->getOutput(0); - - } else { - ind = ind < 0 ? ind + in->getDimensions().d[dim] : ind; - // index to access needs to be an at::Tensor - at::Tensor indices = torch::tensor({ind}).to(torch::kI32); - index_tensor = tensor_to_const(engine, indices); - } - } else { - POROS_CHECK_TRUE((in->getDimensions().d[dim] >= 0), "When index(input[2]) of aten::select is not from prim::Constant," - " the selected " + std::to_string(dim) + "th dim of input must be fixed (not dynamic)." << node_info(node)); - } - - // IGatherLayer takes in input tensor, the indices, and the axis - // of input tensor to take indices from - auto gather_layer = engine->network()->addGather(*in, *index_tensor, dim); - POROS_CHECK(gather_layer, "Unable to create gather layer from node: " << *node); - gather_layer->setName((layer_info(node) + "_gathier").c_str()); - auto out = gather_layer->getOutput(0); - LOG(INFO) << "Gather tensor shape: " << out->getDimensions(); - - if (out->getDimensions().nbDims != 1) { - // IShuffleLayer removes redundant dimensions - auto shuffle_layer = engine->network()->addShuffle(*out); - POROS_CHECK(shuffle_layer, "Unable to create shuffle layer from node: " << *node); - // when input is dynamic - if (check_nvtensor_is_dynamic(out)) { - nvinfer1::ITensor* gather_out_shape_tensor = engine->network()->addShape(*out)->getOutput(0); - gather_out_shape_tensor = squeeze_nv_shapetensor(engine, gather_out_shape_tensor, dim); - shuffle_layer->setInput(1, *gather_out_shape_tensor); - } else { - // when input is not dynamic - shuffle_layer->setReshapeDimensions(squeeze_dims(out->getDimensions(), dim, false)); - } - shuffle_layer->setName(layer_info(node).c_str()); - out = shuffle_layer->getOutput(0); - } - - engine->context().set_tensor(node->outputs()[0], out); - LOG(INFO) << "Output tensor shape: " << out->getDimensions(); - return true; -} - -// aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) -bool SliceConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - - at::ArrayRef inputs = node->inputs(); - - if (node->schema().operator_name() == torch::jit::parseSchema(this->schema_string()[1]).operator_name()) { - // aten::slice.t(t[] l, int? start=None, int? end=None, int step=1) -> (t[]) - POROS_CHECK_TRUE((inputs.size() == 4), "invaid inputs size for SliceConverter"); - - nvinfer1::ITensor* self_nvtensor = nullptr; - std::vector self_vec = {}; - int32_t dim_rank = 0; - std::vector itensor_vec = {}; - bool has_tensor_scalar = false; - // input[0] is int[] - if (inputs[0]->type()->isSubtypeOf(c10::ListType::ofInts())) { - has_tensor_scalar = check_inputs_tensor_scalar(engine, node); - if (has_tensor_scalar) { - self_nvtensor = this->get_tensor_scalar(inputs[0]); - POROS_CHECK_TRUE((self_nvtensor != nullptr), node_info(node) + std::string("get int nvtensor false.")); - dim_rank = (self_nvtensor->getDimensions()).d[0]; - } else { - self_vec = (engine->context().get_constant(inputs[0])).toIntList().vec(); - dim_rank = self_vec.size(); - } - // tensor[] - } else if (inputs[0]->type()->isSubtypeOf(c10::ListType::ofTensors())) { - POROS_CHECK_TRUE(engine->context().get_tensorlist(inputs[0], itensor_vec), "extract tensor list error."); - dim_rank = itensor_vec.size(); - } else { - LOG(WARNING) << node->schema().operator_name() << " converter input[0] meets unsupported type."; - return false; - } - // extract start, end and step - torch::jit::IValue maybe_start = engine->context().get_constant(inputs[1]); - int64_t startIdx = maybe_start.isNone() ? 0 : maybe_start.toInt(); - startIdx = (startIdx < 0) ? (dim_rank + startIdx) : startIdx; - - torch::jit::IValue maybe_end = engine->context().get_constant(inputs[2]); - int64_t endIdx = maybe_end.isNone() ? dim_rank : maybe_end.toInt(); - endIdx = (endIdx < 0) ? (dim_rank + endIdx) : endIdx; - - int64_t step = (engine->context().get_constant(inputs[3])).toInt(); - - POROS_CHECK_TRUE((startIdx <= endIdx && endIdx <= dim_rank), - node_info(node) + std::string("start > end or end > self_size")); - // input[0] is int[] - if (inputs[0]->type()->isSubtypeOf(c10::ListType::ofInts())) { - if (has_tensor_scalar) { - int64_t size = ceil(float(endIdx - startIdx) / float(step)); - std::vector start_vec{startIdx}, size_vec{size}, stride_vec{step}; - auto slice_layer = engine->network()->addSlice(*self_nvtensor, - sizes_to_nvdim(start_vec), - sizes_to_nvdim(size_vec), - sizes_to_nvdim(stride_vec)); - POROS_CHECK(slice_layer, "Unable to given dim info from node: " << *node); - slice_layer->setName(layer_info(node).c_str()); - nvinfer1::ITensor* slice_output = slice_layer->getOutput(0); - engine->context().set_tensor(node->outputs()[0], slice_output); - } else { - c10::List list; - int index = startIdx; - while (index <= endIdx - 1) { - list.push_back(std::move(self_vec[index])); - index += step; - } - auto output_ivalue = c10::optional(std::move(torch::jit::IValue(list))); - engine->context().set_constant(node->outputs()[0], output_ivalue); - } - } else if (inputs[0]->type()->isSubtypeOf(c10::ListType::ofTensors())) { - std::vector output_itensor_vec = {}; - int index = startIdx; - while (index <= endIdx - 1) { - output_itensor_vec.push_back(itensor_vec[index]); - index += step; - } - engine->context().set_tensorlist(node->outputs()[0], output_itensor_vec); - } else { - LOG(WARNING) << node->schema().operator_name() << " converter input[0] meets unsupported type."; - return false; - } - - return true; - } - - POROS_CHECK_TRUE((inputs.size() == 5), "invaid inputs size for SliceConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for SliceConverter is not Tensor as expected"); - for (int32_t i = 1; i < 5; i++) { - if (i == 2 || i == 3) { - continue; - } - POROS_CHECK_TRUE((inputs[i]->node()->kind() == torch::jit::prim::Constant), - std::string("input[") + std::to_string(i) + std::string("] for SliceConverter is not come from prim::Constant as expected")); - } - - nvinfer1::ITensor* in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - - int64_t dim = (engine->context().get_constant(inputs[1])).toInt(); - nvinfer1::Dims in_dims = in->getDimensions(); - int64_t axis = c10::maybe_wrap_dim(dim, in_dims.nbDims); - - torch::jit::IValue maybe_start = engine->context().get_constant(inputs[2]); - int64_t startIdx = maybe_start.isNone() ? 0 : maybe_start.toInt(); - torch::jit::IValue maybe_end = engine->context().get_constant(inputs[3]); - int64_t endIdx = maybe_end.isNone() ? INT64_MAX : maybe_end.toInt(); - int64_t step =(engine->context().get_constant(inputs[4])).toInt(); - POROS_CHECK_TRUE((step > 0), "step for SliceConverter must be postive"); - - int64_t maxDim = static_cast(in_dims.d[axis]); - int64_t start = 0, end = INT64_MAX; - - // not dynamic or axis dim is not negtive - // make sure start and end are postive - if (maxDim >= 0) { - //extract start - start = (startIdx < 0) ? (maxDim + startIdx) : startIdx; - POROS_CHECK_TRUE((start >= 0 && start <= maxDim), "invalid start for SliceConverter"); - //extract end - endIdx = std::min(endIdx, maxDim); - end = (endIdx < 0) ? (maxDim + endIdx) : endIdx; - POROS_CHECK_TRUE((end >= start && end <= maxDim), "invalid end for SliceConverter or end less than start"); - POROS_CHECK_TRUE((step <= maxDim), "invalid step for SliceConverter"); - } - - std::vector start_vec, size_vec, stride_vec; - bool is_dynamic = check_nvtensor_is_dynamic(in); - bool has_tensor_scalar = check_inputs_tensor_scalar(engine, node); - for (int32_t i = 0; i < in_dims.nbDims; i++) { - start_vec.push_back(0); - size_vec.push_back(in_dims.d[i]); - stride_vec.push_back(1); - } - stride_vec[axis] = step; - start_vec[axis] = start; - - nvinfer1::ILayer* slice_layer = nullptr; - - // no dynamic and ints don't have nvtensor inputs. - if (!is_dynamic && !has_tensor_scalar) { - int64_t size = ceil(float(end - start) / float(step)); - size_vec[axis] = size; - slice_layer = engine->network()->addSlice(*in, - sizes_to_nvdim(start_vec), - sizes_to_nvdim(size_vec), - sizes_to_nvdim(stride_vec)); - slice_layer->setName(layer_info(node).c_str()); - } else { // dynamic - nvinfer1::IShapeLayer* shape_layer = engine->network()->addShape(*in); - nvinfer1::ITensor* in_shape_tensor = shape_layer->getOutput(0); - nvinfer1::ITensor* start_tensor = nullptr, *size_tensor = nullptr, *end_tensor = nullptr; - - std::vector dy_mask_vec, dy_rev_mask_vec; - - for (int32_t i = 0; i < in_dims.nbDims; i++) { - dy_mask_vec.push_back(0); - dy_rev_mask_vec.push_back(1); - } - - // Prepare for following calculations. - // Such as, get dynamic input dims is [4, 5, *, 7] (runtime input dims is [4, 5, 6, 7]), and axis dim is 2. - // Then, mask_tensor is [0, 0, 1, 0], rev_mask_tensor is [1, 1, 0, 1], - // mask_shape_tensor is [0, 0, 6, 0], rev_mask_shape_tensor is [4, 5, 0, 7]. - at::Tensor mask_tensor = torch::tensor(dy_mask_vec, torch::kInt); - at::Tensor rev_mask_tensor = torch::tensor(dy_rev_mask_vec, torch::kInt); - - rev_mask_tensor[axis] = 0; - nvinfer1::ITensor* const_rev_mask_tensor = tensor_to_const(engine, rev_mask_tensor); - nvinfer1::ITensor* rev_mask_shape_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - in_shape_tensor, - const_rev_mask_tensor, - layer_info(node) + std::string("_axis_dim_to_zero"))->getOutput(0); - mask_tensor[axis] = 1; - nvinfer1::ITensor* const_mask_tensor = tensor_to_const(engine, mask_tensor); - nvinfer1::ITensor* mask_shape_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - in_shape_tensor, - const_mask_tensor, - layer_info(node) + std::string("_other_dims_to_zero"))->getOutput(0); - bool has_tensor_scalar = check_inputs_tensor_scalar(engine, node); - if (has_tensor_scalar) { - // Generally, only start and end come from nvtensor - // nvinfer1::ITensor* dim_int_nvtensor = this->get_tensor_scalar(inputs[1]); - nvinfer1::ITensor* start_int_nvtensor = this->get_tensor_scalar(inputs[2]); - nvinfer1::ITensor* end_int_nvtensor = this->get_tensor_scalar(inputs[3]); - // nvinfer1::ITensor* stride_int_nvtensor = this->get_tensor_scalar(inputs[4]); - - // only end from nvtensor (start is none) - if (end_int_nvtensor != nullptr && start_int_nvtensor == nullptr) { - LOG(INFO) << "Slice only end from nvtensor"; - mask_tensor[axis] = 0; - start_tensor = tensor_to_const(engine, mask_tensor); - nvinfer1::ITensor* end_tensor_temp = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - const_mask_tensor, - end_int_nvtensor, - layer_info(node) + std::string("_end_prod_mask_shape_tensor"))->getOutput(0); - end_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - end_tensor_temp, - rev_mask_shape_tensor, - layer_info(node) + std::string("_end_tmp_sum_rev_mask_shape_tensor"))->getOutput(0); - // only start from nvtensor (end is none) - } else if (end_int_nvtensor == nullptr && start_int_nvtensor != nullptr) { - LOG(INFO) << "Slice only start from nvtensor"; - start_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - const_mask_tensor, - start_int_nvtensor, - layer_info(node) + std::string("_start_prod_mask_shape_tensor"))->getOutput(0); - end_tensor = in_shape_tensor; - // start and end both from nvtensor - } else { - LOG(INFO) << "Slice start and end both from nvtensor"; - // make sure that start or end which not from nvtensor is postive when maxDims >= 0 - if (maxDim >= 0) { - if (!maybe_start.isNone()) { - LOG(INFO) << "Slice start can be from constant"; - start_int_nvtensor = tensor_to_const(engine, torch::tensor({start}, torch::kInt)); - } - if (!maybe_end.isNone()) { - LOG(INFO) << "Slice end can be from constant"; - end_int_nvtensor = tensor_to_const(engine, torch::tensor({end}, torch::kInt)); - } - } - start_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - const_mask_tensor, - start_int_nvtensor, - layer_info(node) + std::string("_start_prod_mask_shape_tensor"))->getOutput(0); - nvinfer1::ITensor* end_tensor_temp = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - const_mask_tensor, - end_int_nvtensor, - layer_info(node) + std::string("_end_prod_mask_shape_tensor"))->getOutput(0); - end_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - end_tensor_temp, - rev_mask_shape_tensor, - layer_info(node) + std::string("_end_tmp_sum_rev_mask_shape_tensor"))->getOutput(0); - } - nvinfer1::ITensor* sub_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUB, - end_tensor, - start_tensor, - layer_info(node) + std::string("_end_sub_start"))->getOutput(0); - // Equivalent to ceil((end - start) / step) -> size - if (step > 1) { - mask_tensor[axis] = step - 1; - nvinfer1::ITensor* sum_step_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - sub_tensor, - tensor_to_const(engine, mask_tensor), - layer_info(node) + std::string("_sum_step_sub_one"))->getOutput(0); - rev_mask_tensor[axis] = step; - size_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kFLOOR_DIV, - sum_step_tensor, - tensor_to_const(engine, rev_mask_tensor), - layer_info(node) + std::string("_div_get_size"))->getOutput(0); - } else { - size_tensor = sub_tensor; - } - - } else { - if (maxDim < 0) { - // start - mask_tensor[axis] = startIdx; - if (startIdx < 0) { - start_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - mask_shape_tensor, - tensor_to_const(engine, mask_tensor), - layer_info(node) + std::string("_start_tensor"))->getOutput(0); - } else { - start_tensor = tensor_to_const(engine, mask_tensor); - } - // end - if (maybe_end.isNone()){ - end_tensor = in_shape_tensor; - } else { - mask_tensor[axis] = endIdx; - if (endIdx < 0) { - end_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - in_shape_tensor, - tensor_to_const(engine, mask_tensor), - layer_info(node) + std::string("_end_tensor_to_pos"))->getOutput(0); - } else { - end_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - rev_mask_shape_tensor, - tensor_to_const(engine, mask_tensor), - layer_info(node) + std::string("_end_tensor"))->getOutput(0); - } - } - nvinfer1::ITensor* sub_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUB, - end_tensor, - start_tensor, - layer_info(node) + std::string("_end_sub_start"))->getOutput(0); - // Equivalent to ceil((end - start) / step) -> size - if (step > 1) { - mask_tensor[axis] = step - 1; - nvinfer1::ITensor* sum_step_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - sub_tensor, - tensor_to_const(engine, mask_tensor), - layer_info(node) + std::string("_sum_step_sub_one"))->getOutput(0); - rev_mask_tensor[axis] = step; - size_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kFLOOR_DIV, - sum_step_tensor, - tensor_to_const(engine, rev_mask_tensor), - layer_info(node) + std::string("_div_get_size"))->getOutput(0); - } else { - size_tensor = sub_tensor; - } - } else { - mask_tensor[axis] = start; - start_tensor = tensor_to_const(engine, mask_tensor); - - mask_tensor[axis] = ceil(float(end - start) / float(step)); - size_tensor = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kSUM, - rev_mask_shape_tensor, - tensor_to_const(engine, mask_tensor), - layer_info(node) + std::string("_sum_get_size"))->getOutput(0); - } - } - - std::vector temp_vec = {0, 0}; - slice_layer = engine->network()->addSlice(*in, sizes_to_nvdim(temp_vec), - sizes_to_nvdim(temp_vec), - sizes_to_nvdim(stride_vec)); - slice_layer->setInput(0, *in); - slice_layer->setInput(1, *start_tensor); - slice_layer->setInput(2, *size_tensor); - // slice_layer->setInput(3, *stride_tensor); - slice_layer->setName(layer_info(node).c_str()); - } - - nvinfer1::ITensor* slice_out = slice_layer->getOutput(0); - engine->context().set_tensor(node->outputs()[0], slice_out); - LOG(INFO) << "Output tensor shape: " << slice_out->getDimensions(); - return true; -} - -/*aten::embedding(Tensor weight, -Tensor indices, -int padding_idx=-1, -bool scale_grad_by_freq=False, -bool sparse=False) -> Tensor*/ -bool EmbeddingConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 5), "invaid inputs size for EmbeddingConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for EmbeddingConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[1]->type()->isSubtypeOf(c10::TensorType::get())), - "input[1] for EmbeddingConverter is not Tensor as expected"); - - auto embedding = engine->context().get_tensor(inputs[0]); - auto indices = engine->context().get_tensor(inputs[1]); - POROS_CHECK_TRUE(((embedding != nullptr) && (indices != nullptr)), - "Unable to init input tensor for node: " << *node); - - // Set datatype for indices tensor to INT32 - auto identity = engine->network()->addIdentity(*indices); - identity->setOutputType(0, nvinfer1::DataType::kINT32); - identity->setName((layer_info(node) + "_identify").c_str()); - indices = identity->getOutput(0); - - // IGatherLayer takes in input tensor, the indices, and the axis of input tensor to take indices from - auto gather_layer = engine->network()->addGather(*embedding, *indices, 0); - POROS_CHECK(gather_layer, "Unable to create gather layer from node: " << *node); - gather_layer->setName(layer_info(node).c_str()); - auto gather_out = gather_layer->getOutput(0); - - engine->context().set_tensor(node->outputs()[0], gather_out); - LOG(INFO) << "Output tensor shape: " << gather_out->getDimensions(); - return true; -} - -/* -aten::narrow(Tensor(a) self, int dim, int start, int length) -> Tensor(a) -aten::narrow.Tensor(Tensor(a) self, int dim, Tensor start, int length) -> Tensor(a) -*/ -bool NarrowConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 4), "invaid inputs size for NarrowConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for NarrowConverter is not Tensor as expected"); - - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - - //extract dim & length - auto maxDim = static_cast(in->getDimensions().nbDims); - auto axis = (engine->context().get_constant(inputs[1])).toInt(); - axis = (axis < 0) ? (axis + maxDim) : axis; - auto length = (int32_t)(engine->context().get_constant(inputs[3])).toInt(); - - //extract start - int32_t start = 0; - auto maybe_start = engine->context().get_constant(inputs[2]); - if (maybe_start.isInt()) { - start = (int32_t)maybe_start.toInt(); - start = (start < 0) ? (maxDim + start) : start; - } else if (maybe_start.isTensor()) { - auto start_tensor = maybe_start.toTensor().to(torch::kI32); - start = start_tensor.item().to(); - } - - // index to access needs to be an at::Tensor - at::Tensor indices = torch::arange(start, start + length, 1).to(torch::kI32); - auto weights = Weights(indices); - - // IConstantLayer to convert indices from Weights to ITensor - auto const_layer = engine->network()->addConstant(weights.shape, weights.data); - POROS_CHECK(const_layer, "Unable to create constant layer from node: " << *node); - auto const_out = const_layer->getOutput(0); - - // IGatherLayer takes in input tensor, the indices, and the axis - // of input tensor to take indices from - auto gather_layer = engine->network()->addGather(*in, *const_out, axis); - POROS_CHECK(gather_layer, "Unable to create gather layer from node: " << *node); - auto gather_out = gather_layer->getOutput(0); - - // IShuffleLayer removes redundant dimensions - auto shuffle_layer = engine->network()->addShuffle(*gather_out); - POROS_CHECK(shuffle_layer, "Unable to create shuffle layer from node: " << *node); - shuffle_layer->setReshapeDimensions(unpad_nvdim(gather_out->getDimensions())); - shuffle_layer->setName(layer_info(node).c_str()); - auto shuffle_out = shuffle_layer->getOutput(0); - engine->context().set_tensor(node->outputs()[0], shuffle_out); - LOG(INFO) << "Output tensor shape: " << shuffle_out->getDimensions(); - return true; -} - -/* -aten::split.Tensor(Tensor(a) self, int split_size, int dim=0) -> Tensor(a)[] -aten::split_with_sizes(Tensor(a) self, int[] split_sizes, int dim=0) -> Tensor(a)[] -aten::unbind.int(Tensor(a) self, int dim=0) -> Tensor(a)[] -*/ -bool SplitConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 3 || inputs.size() == 2), "invaid inputs size for SplitConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for SplitConverter is not Tensor as expected"); - - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - - int axis = 0; - //extract dim - if (inputs.size() == 3) { - axis = (engine->context().get_constant(inputs[2])).toInt(); - } else { - // node->kind() == torch::jit::aten::unbind - // aten::unbind 和 split_size=1时的aten::split 非常像。但aten::unbind最后会做一次squeeze。 - // 例如:输入shape(2,3,4),dim=1,split_size=1, - // 那么aten::unbind出来的就是3个(2,4),aten::split出来的就是3个(2,1,4) - axis = (engine->context().get_constant(inputs[1])).toInt(); - } - - auto in_dim_size = in->getDimensions().d[axis]; - - //extract split_size - auto num_outputs = 1; - auto num_remainder = 0; - std::vector sizes; - auto maybe_split_size = engine->context().get_constant(inputs[1]); - if (node->kind() == torch::jit::aten::split_with_sizes) { - sizes = maybe_split_size.toIntList().vec(); - num_outputs = sizes.size(); - } else { // node->kind() == torch::jit::aten::split - auto split_size = maybe_split_size.toInt(); - // node->kind() == torch::jit::aten::unbind 时设置 split_size 为 1 - if (inputs.size() == 2) { - split_size = 1; - } - num_outputs = in_dim_size / split_size; - num_remainder = in_dim_size % split_size; - for (int64_t i = 0; i < num_outputs; i++) { - sizes.push_back(split_size); - } - if (num_remainder) { - num_outputs += 1; - sizes.push_back(num_remainder); - } - } - - LOG(INFO) << "Number of split outputs: " << num_outputs; - - std::vector tensorlist; - tensorlist.reserve(num_outputs); - - int start_idx = 0; - for (int64_t i = 0; i < num_outputs; i++) { - at::Tensor indices = torch::arange(start_idx, start_idx + sizes[i], 1).to(torch::kI32); - auto indices_tensor = tensor_to_const(engine, indices); - - auto gather_layer = engine->network()->addGather(*in, *indices_tensor, axis); - auto gather_out = gather_layer->getOutput(0); - // 为 aten::unbind axis维度做一次 squeeze - if (inputs.size() == 2) { - nvinfer1::IShuffleLayer* shuffle_l = engine->network()->addShuffle(*gather_out); - std::vector in_shape_vec = nvdim_to_sizes(in->getDimensions()); - in_shape_vec.erase(in_shape_vec.begin() + axis); - shuffle_l->setReshapeDimensions(sizes_to_nvdim(in_shape_vec)); - gather_out = shuffle_l->getOutput(0); - } - - tensorlist.emplace_back(gather_out); - start_idx = start_idx + sizes[i]; - } - - engine->context().set_tensorlist(node->outputs()[0], tensorlist); - return true; -} - -/* -aten::masked_fill.Scalar(Tensor self, Tensor mask, Scalar value) -> Tensor -aten::masked_fill.Tensor(Tensor self, Tensor mask, Tensor value) -> Tensor -*/ -bool MaskedFillConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 3), "invaid inputs size for MaskedFillConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for MaskedFillConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[1]->type()->isSubtypeOf(c10::TensorType::get())), - "input[1] for MaskedFillConverter is not Tensor as expected"); - - //extract self & mask - auto self = engine->context().get_tensor(inputs[0]); - auto mask = engine->context().get_tensor(inputs[1]); - POROS_CHECK_TRUE((self != nullptr && mask != nullptr), "Unable to init input tensor for node: " << *node); - int max_rank = std::max({self->getDimensions().nbDims, mask->getDimensions().nbDims}); - - bool is_dynamic = check_nvtensor_is_dynamic(self) || check_nvtensor_is_dynamic(mask); - if (is_dynamic) { - self = broadcast_itensor(engine, node, self, max_rank, "self"); - mask = broadcast_itensor(engine, node, mask, max_rank, "mask"); - } else { - mask = add_padding(engine, node, mask, max_rank, false, true); - self = add_padding(engine, node, self, max_rank, false, true); - } - - //extract value - nvinfer1::ITensor* val_t = engine->context().get_tensor(inputs[2]); - //situation1: val is a scalar and is_dynamic == false - if (val_t == nullptr && !is_dynamic) { - auto val = (engine->context().get_constant(inputs[2])).toScalar().to(); - val_t = tensor_to_const(engine, torch::full(nvdim_to_sizes(self->getDimensions()), val)); - //situation2: val is a scalar and is_dynamic == true - } else if (val_t == nullptr && is_dynamic) { - //change scalar to tensor and broadcast it - auto val = (engine->context().get_constant(inputs[2])).toScalar().to(); - at::Tensor val_at_tensor = torch::tensor({val}); - nvinfer1::ITensor* val_nv_tensor = tensor_to_const(engine, val_at_tensor); - val_t = broadcast_itensor(engine, node, val_nv_tensor, max_rank, "value"); - //situation3: val is a tensor - } else { - int32_t value_rank = val_t->getDimensions().nbDims; - POROS_CHECK(value_rank == 0, "masked_fill only supports a 0-dimensional value tensor"); - //let's expand value - int32_t new_value_rank = self->getDimensions().nbDims; - //nvinfer1::ITensor* new_value_shape = engine->network()->addShape(*self)->getOutput(0); - - //先给value把维度补起来,补成[1, 1, 1, ...], 用shuffle实现 - std::vector new_dim(new_value_rank, 1); - auto reshape_layer = engine->network()->addShuffle(*val_t); - reshape_layer->setReshapeDimensions(sizes_to_nvdim(c10::IntArrayRef(new_dim))); - reshape_layer->setName((layer_info(node) + "_IShuffleLayer_for_value").c_str()); - val_t = reshape_layer->getOutput(0); - - //无需专门调用slice把维度对齐,addSelect接口要求rank对齐就行,rank对齐的情况下,接口内部自己会broadcast。 - /* - //再slice一下, 因为是从rank 0 expand到其他的dim, - //所以此处start_dim 设置为全0,stride_dim 也设置为全0, - //sizes信息先用start_dim 作为dummy input, 后面用setInput 接口设置真是的output_dim 信息。 - std::vector start_vec_new(new_value_rank, 0); - auto offset = sizes_to_nvdim(c10::IntArrayRef(start_vec_new)); - - // Slice layer does the expansion in TRT. Desired output size is specified by new_value_shape - auto slice_layer = engine->network()->addSlice(*val_t, offset, offset, offset); - slice_layer->setInput(2, *new_value_shape); - slice_layer->setName((layer_info(node) + "_ISliceLayer_for_value").c_str()); - val_t = slice_layer->getOutput(0); - */ - } - - //no need anymore - // POROS_CHECK(broadcastable(self->getDimensions(), mask->getDimensions(), /*multidirectional=*/false), - // "Self and mask tensors are not broadcastable"); - - nvinfer1::ISelectLayer* new_layer = engine->network()->addSelect(*mask, *val_t, *self); - POROS_CHECK(new_layer, "Unable to create layer for aten::masked_fill"); - - new_layer->setName(layer_info(node).c_str()); - engine->context().set_tensor(node->outputs()[0], new_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << new_layer->getOutput(0)->getDimensions(); - return true; -} - -// aten::gather(Tensor self, int dim, Tensor index, *, bool sparse_grad=False) -> Tensor -bool GatherConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 4), "invaid inputs size for GatherConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for GatherConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[1]->node()->kind() == torch::jit::prim::Constant), - "input[1] for GatherConverter is not come from prim::Constant as expected"); - POROS_CHECK_TRUE((inputs[2]->type()->isSubtypeOf(c10::TensorType::get())), - "input[2] for GatherConverter is not Tensor as expected"); - // extract self - nvinfer1::ITensor* self = engine->context().get_tensor(inputs[0]); - auto maxDim = static_cast(self->getDimensions().nbDims); - // extract index - nvinfer1::ITensor* index = engine->context().get_tensor(inputs[2]); - POROS_CHECK_TRUE(((self != nullptr) && (index != nullptr)), - "Unable to init input tensor for node: " << *node); - //extract dim - int64_t dim = engine->context().get_constant(inputs[1]).toInt(); - // make sure dim >= 0 - dim = dim < 0 ? dim + maxDim : dim; - - // Set datatype for indices tensor to INT32 - nvinfer1::IIdentityLayer* identity = engine->network()->addIdentity(*index); - identity->setOutputType(0, nvinfer1::DataType::kINT32); - identity->setName((layer_info(node) + "_identify").c_str()); - index = identity->getOutput(0); - - // IGatherLayer takes in input tensor, the indices, and the axis of input tensor to take indices from - nvinfer1::IGatherLayer* gather_layer = engine->network()->addGather(*self, *index, dim); - POROS_CHECK(gather_layer, "Unable to create gather layer from node: " << *node); - gather_layer->setName(layer_info(node).c_str()); - gather_layer->setMode(nvinfer1::GatherMode::kELEMENT); - nvinfer1::ITensor* gather_out = gather_layer->getOutput(0); - - engine->context().set_tensor(node->outputs()[0], gather_out); - LOG(INFO) << "Output tensor shape: " << gather_out->getDimensions(); - return true; -} - -/* -aten::index含义:用indices指定下标,选取self指定维度。 -(实际上是用tensor将多个indices分dims打包起来,能够一起选取) -例如:输入x,其shape = {3, 4, 5} -输入两个indices tensors, -indices_1 = [0, 2] -indices_2 = [1, 3] -这组输入表示用indices_1选取x dim=0 的 0和2下标,用indices_2选取x dim=1 的 1和3下标 -则结果为 -output = [x[0][1], x[2][3]] -由于剩余x dim=2 的维度是5,则 -output.shape = {1, 2, 5} -这样就实现了同时选取 x[0][1] 和 x[2][3] 的功能了。 -规则: -1、输入的indices数量不能超过self rank数。(也就是说本例子中输入的indices tensor数量不能大于3) -2、输入的indices中的值不能超过自己对应维度范围。(例如:indices_1对应x的dim=0,则其最大值必须小于3;indices_2对应x的dim=1,则其最大值必须小于4。) -3、输入的indices shape必须一致或可以broadcast。(为的是相应位置能够同时选取。) ---------------------------------- -如果继续上面再输入一个indices tensor -indices_3 = [2, 4] -则结果为 -output = [x[0][1][2], x[2][3][4]] -output.shape = {1, 2} -*/ -// aten::index.Tensor(Tensor self, Tensor?[] indices) -> Tensor -bool IndexConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for IndexConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for IndexConverter is not Tensor as expected"); - // torch对于Tensor?[]类型的解释: - // the index of aten::index should be a type of List[Optional[Tensor]], - // this is to support the case like t[:, :, 1] where : here indicates a - // None/undefined tensor(optional tensor) - POROS_CHECK_TRUE((inputs[1]->type()->str().find("Tensor?[]") != std::string::npos), - "input[1] for IndexConverter is not List[Optional[Tensor]] (Tensor?[]) as expected"); - - // extract self - nvinfer1::ITensor* self = engine->context().get_tensor(inputs[0]); - // extract indices - std::vector indices_tensors; - engine->context().get_tensorlist(inputs[1], indices_tensors); - - // ps: 目前能支持在self的0 dim上选取,也就是说indices_tensors只能输入一个。 - // 而下面注释掉这段代码实现了更全面的功能,支持多个indices_tensors(见下方介绍),但是实测中会使模型速度更慢(由于使用了更多的gather),先注释掉。解掉注释不支持dynamic - POROS_CHECK_TRUE((indices_tensors.size() == 1), - "aten::Index of torchscript implements the selection of multiple dimensions with several indices. " - "But due to the functional limitations of trt gatherlayer, in this version of poros, " - "aten::Index only support one indice input, which means only 0 dim of self can be indexed."); - - /* - // 设self.rank = r,indices_tensors.size() = q (根据规则1 q <= r),则该组输入会选取self的前q维。 - // 现由于trt gatherlayer功能限制,现只能实现self前q-1维的每一维只能选取一个值。 - // 例如:上面例子 output = [x[0][1][2], x[2][3][4]] 是不能支持的(因为选取的第1维同时有0和2,第2维同时有1和3), - // 而output = [x[0][1][2], x[0][1][4]]是能够支持的。(因为选取的第1维只有0,第2维只有1) - // 换句话说,第1至q-1的indices_tensors中的每个值都必须相等。 - // 为便于判断,先设定只有前q-1的indices_tensors所有维度都是1才能支持(因为这样broadcast过去能保证indices_tensor中的每个值都相等) - for (size_t i = 0; i < indices_tensors.size() - 1; i++) { - std::vector input_index_shape_vec = nvdim_to_sizes(indices_tensors[i]->getDimensions()); - size_t shape_prod = 1; - for (size_t j = 0; j < input_index_shape_vec.size(); j++) { - shape_prod *= input_index_shape_vec[j]; - } - if (shape_prod > 1) { - LOG(WARNING) << "Torchscript could have implemented aten::Index with several indices. But due to the functional limitations of trt gatherlayer, " - "in this version of poros, aten::Index only support that every dimension of indices is equal to 1 except the last one."; - return false; - } - } - // 前q - 1维选取 - for (size_t i = 0; i < indices_tensors.size() - 1; i++) { - // Set datatype for indices tensor to INT32 - nvinfer1::IIdentityLayer* identity_layer = engine->network()->addIdentity(*indices_tensors[i]); - identity_layer->setOutputType(0, nvinfer1::DataType::kINT32); - identity_layer->setName((layer_info(node) + "_identify" + std::to_string(i)).c_str()); - indices_tensors[i] = identity_layer->getOutput(0); - - // 由于前q-1的indices_tensors所有维度都是1,可以将indices reshape到1维 - nvinfer1::IShuffleLayer* shuffle_layer = engine->network()->addShuffle(*indices_tensors[i]); - POROS_CHECK(shuffle_layer, "Unable to create shuffle layer from node: " << *node); - shuffle_layer->setName((layer_info(node) + "_shuffle" + std::to_string(i)).c_str()); - std::vector one_vec = {1}; - shuffle_layer->setReshapeDimensions(sizes_to_nvdim(one_vec)); - indices_tensors[i] = shuffle_layer->getOutput(0); - - // 用1维的indices 去gather self的第0维 - nvinfer1::IGatherLayer* gather_layer = engine->network()->addGather(*self, *indices_tensors[i], 0); - POROS_CHECK(gather_layer, "Unable to create gather layer from node: " << *node); - gather_layer->setName((layer_info(node) + "_gather" + std::to_string(i)).c_str()); - self = gather_layer->getOutput(0); - - // 由于gather出的结果第0维是1,可以将gather出的第0维抹掉 - auto self_shape_vec = nvdim_to_sizes(self->getDimensions()); - self_shape_vec.erase(self_shape_vec.begin()); - nvinfer1::IShuffleLayer* shuffle_layer2 = engine->network()->addShuffle(*self); - POROS_CHECK(shuffle_layer2, "Unable to create shuffle layer from node: " << *node); - shuffle_layer->setName((layer_info(node) + "_shuffle2_" + std::to_string(i)).c_str()); - shuffle_layer2->setReshapeDimensions(sizes_to_nvdim(self_shape_vec)); - self = shuffle_layer2->getOutput(0); - }*/ - - // 最后一维选取,支持indices中包含多个不同值 - nvinfer1::ITensor* final_index = *(--indices_tensors.end()); - // Set datatype for indices tensor to INT32 - nvinfer1::IIdentityLayer* identity_layer = engine->network()->addIdentity(*final_index); - identity_layer->setOutputType(0, nvinfer1::DataType::kINT32); - identity_layer->setName((layer_info(node) + "_identify").c_str()); - final_index = identity_layer->getOutput(0); - - // IGatherLayer takes in input tensor, the indices, and the axis of input tensor to take indices from - nvinfer1::IGatherLayer* gather_layer = engine->network()->addGather(*self, *final_index, 0); - POROS_CHECK(gather_layer, "Unable to create gather layer from node: " << *node); - gather_layer->setName((layer_info(node) + "_gather").c_str()); - nvinfer1::ITensor* gather_out = gather_layer->getOutput(0); - - engine->context().set_tensor(node->outputs()[0], gather_out); - LOG(INFO) << "Output tensor shape: " << gather_out->getDimensions(); - return true; -} - -//aten::index_put(Tensor self, Tensor?[] indices, Tensor values, bool accumulate=False) -> Tensor -//TODO: when meet accumulate == True situation. not support yet. -//TODO: when indices element type is Bool, not support yet. -bool IndexPutConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 4), "invaid inputs size for IndexPutConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for IndexPutConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[1]->type()->str().find("Tensor?[]") != std::string::npos), - "input[1] for IndexPutConverter is not List[Optional[Tensor]] (Tensor?[]) as expected"); - - // extract self - nvinfer1::ITensor* self = engine->context().get_tensor(inputs[0]); - // extract indices - std::vector indices_tensors; - engine->context().get_tensorlist(inputs[1], indices_tensors); - //extract values - nvinfer1::ITensor* values = engine->context().get_tensor(inputs[2]); - //extract accumulate - bool accumulate = (engine->context().get_constant(inputs[3])).toBool(); - - //situation 1/3: ---------- when indices_tensors.size() == 0 ------------- - if (indices_tensors.size() == 0) { - engine->context().set_tensor(node->outputs()[0], values); - LOG(WARNING) << "meet the situation when indices_tensors(the second input value) for index_put is empty."; - LOG(INFO) << "Output tensor shape: " << values->getDimensions(); - return true; - } - - if (accumulate == true) { - LOG(WARNING) << "accumulate equal true situation is not supported yet"; - return false; - } - - LOG(INFO) << "handle node info: " << node_info(node) - << ", self tensor shape: " << self->getDimensions() - << ", value tensor shape: " << values->getDimensions() - << ", indices_tensors.size(): " << indices_tensors.size(); - - auto is_dynamic_shape = PorosGlobalContext::instance().get_poros_options().is_dynamic; - nvinfer1::ITensor* index_tensor = nullptr; - nvinfer1::ITensor* broadcast_index_shape = nullptr; - //situation 2/3: ---------- when indices_tensors.size() > 1 ------------- - if (indices_tensors.size() > 1) { - //TODO: check the indices type, if scalartype is bool. we should add NonZero handler - - nvinfer1::ITensor* broadcast_index = indices_tensors[0]; - //add the element in tensor_list to get broadcast index_tensor - for (size_t index = 1; index < indices_tensors.size(); index++) { - auto add = add_elementwise(engine, nvinfer1::ElementWiseOperation::kSUM, - broadcast_index, - indices_tensors[index], - layer_info(node) + "select_add_" + std::to_string(index)); - broadcast_index = add->getOutput(0); - } - //get broadcast_index shape. - LOG(INFO) << "broadcast_index dim is : " << broadcast_index->getDimensions(); - broadcast_index_shape = engine->network()->addShape(*broadcast_index)->getOutput(0); //shape tensor - auto target_dims = broadcast_index->getDimensions(); - auto output_rank = target_dims.nbDims; - - std::vector new_indices_tensors; - - nvinfer1::ITensor* new_input_shape_tensor = nullptr; - nvinfer1::ITensor* in = nullptr; //current handle indice tensor - for (size_t index = 0; index < indices_tensors.size(); index++) { - //step 2.0: expand the indices - in = indices_tensors[index]; - auto input_dims = in->getDimensions(); - auto input_rank = in->getDimensions().nbDims; - LOG(INFO) << "try to expand tensor shape: " << in->getDimensions() - << " to new shape: " << broadcast_index->getDimensions() - << ", input rank: " << input_rank << ", output rank: " << output_rank; - //situation1: ---------- when input is dynamic shape ------------- - if (is_dynamic_shape == true) { - size_t max_rank = std::max(input_rank, output_rank); - // Dimensions are right alignment. Eg: an input of [3, 1] and max_rank = 4, the result of concat is [1, 1, 3, 1] - if (max_rank - input_rank > 0) { //need shuffle - torch::Tensor the_one = torch::tensor(std::vector(max_rank - input_rank, 1), torch::kInt32); - auto one_tensor = tensor_to_const(engine, the_one); - auto in_shape_tensor = engine->network()->addShape(*in)->getOutput(0); - nvinfer1::ITensor* const args[2] = {one_tensor, in_shape_tensor}; - new_input_shape_tensor = engine->network()->addConcatenation(args, 2)->getOutput(0); - } else { //max_rank - input_rank == 0 - new_input_shape_tensor = engine->network()->addShape(*in)->getOutput(0); - } - auto shuffle = engine->network()->addShuffle(*in); - shuffle->setInput(1, *new_input_shape_tensor); - //LOG(INFO) << "input shuffle to shape: " << shuffle->getOutput(0)->getDimensions(); - - // Start the slicing from beginning of tensor since this is an expand layer - std::vector start_vec(max_rank, 0); - nvinfer1::Dims starts_dim = sizes_to_nvdim(c10::IntArrayRef(start_vec)); - at::Tensor th_start = torch::tensor(nvdim_to_sizes(starts_dim), torch::kInt32); - auto starts = tensor_to_const(engine, th_start); - - // compute sizes = max(x,y). - auto sizes = engine->network()->addElementWise(*new_input_shape_tensor, - *broadcast_index_shape, - nvinfer1::ElementWiseOperation::kMAX)->getOutput(0); - nvinfer1::Dims sizes_dim{-1, {}}; - sizes_dim.nbDims = max_rank; - - // Compute (x > 1 ? 1 : 0) for x in newDims, assuming positive x, using only TensorRT operations. - // min(1, sub(input_shape, 1)) - torch::Tensor thOne = torch::tensor({1}, torch::kInt32); - auto thone_tensor = tensor_to_const(engine, thOne); - auto x_sub_one = engine->network()->addElementWise(*new_input_shape_tensor, - *thone_tensor, - nvinfer1::ElementWiseOperation::kSUB)->getOutput(0); - auto strides = engine->network()->addElementWise(*thone_tensor, - *x_sub_one, - nvinfer1::ElementWiseOperation::kMIN)->getOutput(0); - nvinfer1::Dims strides_dim{-1, {}}; - strides_dim.nbDims = max_rank; - - // Slice layer does the expansion in TRT. Desired output size is specified by sizes input at index 2. - auto slice = engine->network()->addSlice(*shuffle->getOutput(0), starts_dim, sizes_dim, strides_dim); - slice->setInput(1, *starts); - slice->setInput(2, *sizes); - slice->setInput(3, *strides); - auto new_indice = slice->getOutput(0); - //LOG(INFO) << "new indice tensor shape: " << new_indice->getDimensions(); - - //unsqueeze it. - auto dim = nvdim_to_sizes(new_indice->getDimensions()).size(); //this is ok - auto shuffle_layer = engine->network()->addShuffle(*new_indice); - nvinfer1::ITensor* input_shape_tensor = (engine->network()->addShape(*new_indice))->getOutput(0); - nvinfer1::ITensor* reshape_tensor = unsqueeze_nv_shapetensor(engine, input_shape_tensor, dim); - shuffle_layer->setInput(1, *reshape_tensor); - //LOG(INFO) << "unsqueeze new indice tensor shape: " << shuffle_layer->getOutput(0)->getDimensions(); - - new_indices_tensors.push_back(shuffle_layer->getOutput(0)); - - //situation2: ---------- when input is NOT dynamic shape ------------- - } else { // is_dynamic_shape == false - // Validate the expansion. Eg: an input of [3, 1] can be expanded to [1, 3, 4] but not [3, 4, 1] - for (int64_t i = target_dims.nbDims - 1; i >= 0; --i) { - int64_t offset = target_dims.nbDims - 1 - i; - int64_t dim = input_dims.nbDims - 1 - offset; - int64_t targetSize = target_dims.d[i]; - // In expand layer passing -1 as the size for a dimension means not changing the size of that dimension. - if (targetSize == -1) { - // in(3, 1), expand(3, -1, 4) -> expand(3, 3, 4) - target_dims.d[i] = input_dims.d[dim]; - } - } - - auto num_expand_dims = target_dims.nbDims - input_dims.nbDims; - if (num_expand_dims > 0) { - nvinfer1::Dims reshape_dims; - reshape_dims.nbDims = target_dims.nbDims; - for (int64_t i = 0; i < num_expand_dims; i++) { - reshape_dims.d[i] = 1; - } - for (int64_t i = 0; i < input_dims.nbDims; i++) { - reshape_dims.d[num_expand_dims + i] = input_dims.d[i]; - } - - // Add a reshape layer to expand dims - auto reshape_layer = engine->network()->addShuffle(*in); - reshape_layer->setReshapeDimensions(reshape_dims); - in = reshape_layer->getOutput(0); - //LOG(INFO) << "Input reshaped to : " << in->getDimensions() << " from " << input_dims; - } - - // Start the slicing from beginning of tensor since this is an expand layer - std::vector start_vec(target_dims.nbDims, 0); - auto start_offset = sizes_to_nvdim(c10::IntArrayRef(start_vec)); - - // Set the stride of non singleton dimension to 1 - std::vector strides_vec(target_dims.nbDims, 0); - for (int64_t i = 0; i < target_dims.nbDims; i++) { - strides_vec[i] = (in->getDimensions().d[i] != 1); - } - - auto strides = sizes_to_nvdim(c10::IntArrayRef(strides_vec)); - // Slice layer does the expansion in TRT. Desired output size is specified by target_dims - auto slice_layer = engine->network()->addSlice(*in, start_offset, target_dims, strides); - auto new_indice = slice_layer->getOutput(0); - //LOG(INFO) << "new indice tensor shape: " << new_indice->getDimensions(); - - //unsqueeze it. - auto dim = nvdim_to_sizes(new_indice->getDimensions()).size(); //this is ok - auto shuffle_layer = engine->network()->addShuffle(*new_indice); - shuffle_layer->setReshapeDimensions(unsqueeze_dims(new_indice->getDimensions(), dim)); - //LOG(INFO) << "unsqueeze new indice tensor shape: " << shuffle_layer->getOutput(0)->getDimensions(); - - new_indices_tensors.push_back(shuffle_layer->getOutput(0)); - } - } - - auto dim = new_indices_tensors[0]->getDimensions().nbDims - 1; - auto cat_layer = engine->network()->addConcatenation(new_indices_tensors.data(), new_indices_tensors.size()); - cat_layer->setAxis(static_cast(dim)); - cat_layer->setName((layer_info(node) + "_IConcatenationLayer_for_indices").c_str()); - index_tensor = cat_layer->getOutput(0); - - //situation 3/3: ---------- when indices_tensors.size() == 1 ------------- - } else { - auto indices_tensor = indices_tensors[0]; - broadcast_index_shape = engine->network()->addShape(*indices_tensor)->getOutput(0); - auto dim = nvdim_to_sizes(indices_tensor->getDimensions()).size(); //this is ok - auto shuffle_layer = engine->network()->addShuffle(*indices_tensor); - shuffle_layer->setReshapeDimensions(unsqueeze_dims(indices_tensor->getDimensions(), dim)); - LOG(INFO) << "unsqueeze indice tensor shape: " << shuffle_layer->getOutput(0)->getDimensions(); - index_tensor = shuffle_layer->getOutput(0); - } - - /******************************************************************** - * values handle begin - * ******************************************************************/ - auto value_rank = values->getDimensions().nbDims; - //1 get self shape self_shape is a 1D tensor - nvinfer1::ITensor* self_shape = engine->network()->addShape(*self)->getOutput(0); - nvinfer1::Dims self_shape_dim = self_shape->getDimensions(); - - //2 sub_data_shape = slice(self_shape, axes=[0], starts=[indices_tensors.size()], ends=[INT64_MAX]) - int64_t start = indices_tensors.size(); - int64_t end = static_cast(self_shape_dim.d[0]); - int64_t size = ceil(float(end - start) / float(1)); - - std::vector start_vec = {start}; - std::vector size_vec = {size}; - std::vector stride_vec = {1}; - - nvinfer1::ITensor* sub_data_shape = engine->network()->addSlice(*self_shape, - sizes_to_nvdim(start_vec), - sizes_to_nvdim(size_vec), - sizes_to_nvdim(stride_vec))->getOutput(0); - - //3 values_shape = g.op("Concat", broadcast_index_shape, sub_data_shape, axis_i=0) - std::vector to_concat_tensors = {broadcast_index_shape, sub_data_shape}; - auto shape_cat_layer = engine->network()->addConcatenation(to_concat_tensors.data(), to_concat_tensors.size()); - shape_cat_layer->setName((layer_info(node) + "_IConcatenationLayer_for_values").c_str()); - auto values_shape = shape_cat_layer->getOutput(0); - - //4. we should expand values when it is a singular value - //values = g.op("Expand", values, values_shape) - if (value_rank == 0) { - LOG(INFO) << "given value is rank == 0, expand it now"; - auto new_value_rank = values_shape->getDimensions().d[0]; - - //先给value把维度补起来,补成[1, 1, 1, ...], 用shuffle实现 - std::vector new_dim(new_value_rank, 1); - auto reshape_layer = engine->network()->addShuffle(*values); - reshape_layer->setReshapeDimensions(sizes_to_nvdim(c10::IntArrayRef(new_dim))); - reshape_layer->setName((layer_info(node) + "_IShuffleLayer_for_rank0_values").c_str()); - values = reshape_layer->getOutput(0); - - //再slice一下, 因为是从rank 0 expand到其他的dim, - //所以此处start_dim 设置为全0,stride_dim 也设置为全0, - //sizes信息先用start_dim 作为dummy input, 后面用setInput 接口设置真是的output_dim 信息。 - std::vector start_vec_new(new_value_rank, 0); - auto offset = sizes_to_nvdim(c10::IntArrayRef(start_vec_new)); - - // Slice layer does the expansion in TRT. Desired output size is specified by values_shape - auto slice_layer = engine->network()->addSlice(*values, offset, offset, offset); - slice_layer->setInput(2, *values_shape); - slice_layer->setName((layer_info(node) + "_ISliceLayer_for_rank0_values").c_str()); - values = slice_layer->getOutput(0); - } - - auto reshape_layer_final = engine->network()->addShuffle(*values); - reshape_layer_final->setInput(1, *values_shape); - reshape_layer_final->setName((layer_info(node) + "_IShuffleLayer_for_values").c_str()); - values = reshape_layer_final->getOutput(0); - LOG(INFO) << "new_values tensor shape: " << values->getDimensions(); - /******************************************************************** - * values handle ends - * ******************************************************************/ - - nvinfer1::IScatterLayer* scatter_layer = engine->network()->addScatter(*self, *index_tensor, *values, nvinfer1::ScatterMode::kND); - //scatter_layer->setAxis(0); // no need - scatter_layer->setName((layer_info(node) + "_scatterND").c_str()); - nvinfer1::ITensor* output = scatter_layer->getOutput(0); - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - return true; -} - -// aten::scatter.value(Tensor self, int dim, Tensor index, Scalar value) -> (Tensor) -// For a 3-D tensor, self is updated as: -// self[index[i][j][k]][j][k] = value # if dim == 0 -// self[i][index[i][j][k]][k] = value # if dim == 1 -// self[i][j][index[i][j][k]] = value # if dim == 2 -// ps: self和index的shape不一定一样,所以只遍历index的所有下标。index中不存在的下标self不更新还用原来的值。 -bool ScatterConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 4), "invaid inputs size for ScatterConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for ScatterConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[2]->type()->isSubtypeOf(c10::TensorType::get())), - "input[2] for ScatterConverter is not Tensor as expected"); - - // extract self - nvinfer1::ITensor* self = engine->context().get_tensor(inputs[0]); - // extract dim - int64_t dim = engine->context().get_constant(inputs[1]).toInt(); - auto maxDim = static_cast(self->getDimensions().nbDims); - dim = dim < 0 ? dim + maxDim : dim; - // extract indices - nvinfer1::ITensor* index_tensor = engine->context().get_tensor(inputs[2]); - // extract scalar - auto ivalue_scalar = engine->context().get_constant(inputs[3]); - float scalar = ivalue_scalar.toScalar().to(); - - nvinfer1::DataType self_data_type = self->getType(); - - // IScatterLayer要求输入self必须是float类型 - if (self_data_type != nvinfer1::DataType::kFLOAT) { - auto identity_layer = engine->network()->addIdentity(*self); - identity_layer->setOutputType(0, nvinfer1::DataType::kFLOAT); - identity_layer->setName((layer_info(node) + "_self_identify_float").c_str()); - self = identity_layer->getOutput(0); - } - - // 当value和self类型不一致时,向self对齐。这里手动做一次类型转换对齐精度。 - if (ivalue_scalar.isDouble() && self_data_type == nvinfer1::DataType::kINT32) { - scalar = (float)(int)scalar; - } - - nvinfer1::ITensor* updates_tensor = nullptr; - bool is_dynamic = check_nvtensor_is_dynamic(index_tensor); - - // 输入nvinfer1::IScatterLayer的index和updates的shape必须相同 - if (!is_dynamic) { - std::vector index_dims_vec = nvdim_to_sizes(index_tensor->getDimensions()); - updates_tensor = tensor_to_const(engine, at::full(index_dims_vec, scalar, torch::kFloat32)); - } else { - nvinfer1::ITensor* index_shape_tensor = engine->network()->addShape(*index_tensor)->getOutput(0); - auto fill_layer = engine->network()->addFill(nvinfer1::Dims{1, {1}}, nvinfer1::FillOperation::kLINSPACE); - fill_layer->setInput(0, *index_shape_tensor); - at::Tensor alpha_tensor = torch::tensor(scalar, torch::kFloat32); - fill_layer->setInput(1, *tensor_to_const(engine, alpha_tensor)); // 初始值 - at::Tensor delta_tensor = torch::zeros(index_tensor->getDimensions().nbDims, torch::kFloat32); - fill_layer->setInput(2, *tensor_to_const(engine, delta_tensor)); // delta值 - fill_layer->setName((layer_info(node) + "_fill_index_shape_value").c_str()); - updates_tensor = fill_layer->getOutput(0); - } - - // self tensor data type must be DataType::kFLOAT. - // index tensor data type must be DataType::kINT32. - // updates tensor data type must be DataType::kFLOAT. - nvinfer1::IScatterLayer* scatter_layer = engine->network()->addScatter(*self, *index_tensor, *updates_tensor, nvinfer1::ScatterMode::kELEMENT); - scatter_layer->setAxis(dim); - scatter_layer->setName((layer_info(node) + "_scatter").c_str()); - - nvinfer1::ITensor* output = scatter_layer->getOutput(0); - // 输出不是原来的type需要还原回去,一般是int - if (output->getType() != self_data_type) { - auto identity_layer = engine->network()->addIdentity(*output); - identity_layer->setOutputType(0, self_data_type); - identity_layer->setName((layer_info(node) + "_output_identify_original_type").c_str()); - output = identity_layer->getOutput(0); - } - - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - return true; -} - -// prim::ConstantChunk -bool ChunkConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 1), "invaid inputs size for ChunkConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for ChunkConverter is not Tensor as expected"); - - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - - // In IR, the prim::ConstantChunk always appears in the form of "prim::ConstantChunk[chunks=xx, dim=xx]()". - // And the way to extract its parameters is a little different. - int32_t raw_dim = (int32_t)node->i(torch::jit::attr::dim); - int32_t chunks = (int32_t)node->i(torch::jit::attr::chunks); - - int32_t in_rank = in->getDimensions().nbDims; - // When dim < 0 - raw_dim = raw_dim < 0 ? in_rank + raw_dim : raw_dim; - int32_t in_dim_size = in->getDimensions().d[raw_dim]; - - int32_t every_chunk_size = (int32_t)ceil((double)in_dim_size / (double)chunks); - int32_t remainder_size = in_dim_size % every_chunk_size; - int32_t chunk_num = (int32_t)ceil((double)in_dim_size / (double)every_chunk_size); - - // Check whether the calculated chunk_num is equal to the output_num of the node. - POROS_CHECK_TRUE((chunk_num == (int32_t)(node->outputs().size())), "The caulated chunk_num (" + std::to_string(chunk_num) + - ") is not equal to the node outputs size (" + std::to_string(node->outputs().size()) + ")."); - - std::vector chunk_sizes_vec; - for (int i = 0; i < chunk_num - 1; i++) { - chunk_sizes_vec.push_back(every_chunk_size); - } - if (remainder_size != 0) { - chunk_sizes_vec.push_back(remainder_size); - } else { - chunk_sizes_vec.push_back(every_chunk_size); - } - - std::vector tensorlist; - tensorlist.reserve(chunk_sizes_vec.size()); - - int start_idx = 0; - for (size_t i = 0; i < chunk_sizes_vec.size(); i++) { - at::Tensor indices = torch::arange(start_idx, start_idx + chunk_sizes_vec[i], 1).to(torch::kI32); - auto indices_tensor = tensor_to_const(engine, indices); - - auto gather_layer = engine->network()->addGather(*in, *indices_tensor, raw_dim); - auto gather_out = gather_layer->getOutput(0); - - tensorlist.emplace_back(gather_out); - start_idx = start_idx + chunk_sizes_vec[i]; - } - for (size_t i = 0; i < chunk_sizes_vec.size(); i++) { - engine->context().set_tensor(node->outputs()[i], tensorlist[i]); - LOG(INFO) << "Output tensor shape: " << tensorlist[i]->getDimensions(); - } - - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, SelectConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, SliceConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, EmbeddingConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, NarrowConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, SplitConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, MaskedFillConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, GatherConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, IndexConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, IndexPutConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, ScatterConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, ChunkConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/select.h b/poros/poros/converter/gpu/select.h deleted file mode 100644 index c7a2c5ea3b7..00000000000 --- a/poros/poros/converter/gpu/select.h +++ /dev/null @@ -1,248 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file select.h -* @author tianjinjin@baidu.com -* @date Tue Aug 24 16:31:28 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class SelectConverter : public GpuConverter { -public: - SelectConverter() {} - virtual ~SelectConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::select.int(Tensor(a) self, int dim, int index) -> Tensor(a)"}; - } - - /** TODO: TRY TO SUPPORT SCHEMA PATTERNS BELLOW: - * aten::select.Dimname(Tensor(a) self, Dimname dim, int index) -> Tensor(a) - **/ - const std::vector node_kind() { - return {torch::jit::aten::select}; - } -}; - -class SliceConverter : public GpuConverter { -public: - SliceConverter() {} - virtual ~SliceConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a)", - "aten::slice.t(t[] l, int? start=None, int? end=None, int step=1) -> (t[])" - }; - } - - const std::vector node_kind() { - return {torch::jit::aten::slice}; - } - - bool assign_schema_attr() { - bool result = true; - result &= assign_schema_attr_helper({{"aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a)", {1, 1}}}); - result &= assign_schema_attr_helper({{"aten::slice.t(t[] l, int? start=None, int? end=None, int step=1) -> (t[])", {1, 1}}}); - return result; - } -}; - -class EmbeddingConverter : public GpuConverter { -public: - EmbeddingConverter() {} - virtual ~EmbeddingConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::embedding(Tensor weight, Tensor indices, int padding_idx=-1, bool scale_grad_by_freq=False, bool sparse=False) -> Tensor"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::embedding}; - } -}; - -class NarrowConverter : public GpuConverter { -public: - NarrowConverter() {} - virtual ~NarrowConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::narrow(Tensor(a) self, int dim, int start, int length) -> Tensor(a)", - "aten::narrow.Tensor(Tensor(a) self, int dim, Tensor start, int length) -> Tensor(a)"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::narrow}; - } -}; - -class SplitConverter : public GpuConverter { -public: - SplitConverter() {} - virtual ~SplitConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::split.Tensor(Tensor(a) self, int split_size, int dim=0) -> Tensor(a)[]", - "aten::split_with_sizes(Tensor(a) self, int[] split_sizes, int dim=0) -> Tensor(a)[]", - "aten::unbind.int(Tensor(a) self, int dim=0) -> Tensor(a)[]"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::split, - torch::jit::aten::split_with_sizes, - torch::jit::aten::unbind}; - } - - bool assign_schema_attr() { - bool result = true; - result &= assign_schema_attr_helper({{"aten::split.Tensor(Tensor(a) self, int split_size, int dim=0) -> Tensor(a)[]", {0, 0}}}); - result &= assign_schema_attr_helper({{"aten::split_with_sizes(Tensor(a) self, int[] split_sizes, int dim=0) -> Tensor(a)[]", {0, 0}}}); - result &= assign_schema_attr_helper({{"aten::unbind.int(Tensor(a) self, int dim=0) -> Tensor(a)[]", {0, 0}}}); - return result; - } - -}; - -class MaskedFillConverter : public GpuConverter { -public: - MaskedFillConverter() {} - virtual ~MaskedFillConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::masked_fill.Scalar(Tensor self, Tensor mask, Scalar value) -> Tensor", - "aten::masked_fill.Tensor(Tensor self, Tensor mask, Tensor value) -> Tensor"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::masked_fill}; - } - - bool assign_schema_attr() { - return assign_schema_attr_helper({{"aten::masked_fill.Scalar(Tensor self, Tensor mask, Scalar value) -> Tensor", {1, 0}}}); - } -}; - -class GatherConverter : public GpuConverter { -public: - GatherConverter() {} - virtual ~GatherConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::gather(Tensor self, int dim, Tensor index, *, bool sparse_grad=False) -> Tensor"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::gather}; - } -}; - -class IndexConverter : public GpuConverter { -public: - IndexConverter() {} - virtual ~IndexConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::index.Tensor(Tensor self, Tensor?[] indices) -> Tensor"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::index}; - } -}; - -class IndexPutConverter : public GpuConverter { -public: - IndexPutConverter() {} - virtual ~IndexPutConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::index_put(Tensor self, Tensor?[] indices, Tensor values, bool accumulate=False) -> Tensor"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::index_put}; - } -}; - -class ScatterConverter : public GpuConverter { -public: - ScatterConverter() {} - virtual ~ScatterConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::scatter.value(Tensor self, int dim, Tensor index, Scalar value) -> (Tensor)"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::scatter}; - } -}; - -class ChunkConverter : public GpuConverter { -public: - ChunkConverter() {} - virtual ~ChunkConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"prim::ConstantChunk(...) -> (...)"}; - } - - const std::vector node_kind() { - return {torch::jit::prim::ConstantChunk}; - } - - bool assign_schema_attr() { - return assign_schema_attr_helper({{"prim::ConstantChunk(...) -> (...)", {0, 0}}}); - } -}; -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/shape_handle.cpp b/poros/poros/converter/gpu/shape_handle.cpp deleted file mode 100644 index c8c35d93c3e..00000000000 --- a/poros/poros/converter/gpu/shape_handle.cpp +++ /dev/null @@ -1,158 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file shape_handle.cpp -* @author tianjinjin@baidu.com -* @date Mon Mar 8 11:36:11 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/shape_handle.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/* -"aten::size(Tensor self) -> (int[]) -aten::size.int(Tensor self, int dim) -> int -"*/ -bool AtenSizeConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 1 || inputs.size() == 2), "invaid inputs size for AtenSizeConverter"); - - //extract self - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - - auto shape = engine->network()->addShape(*self); - POROS_CHECK(shape, "Unable to create shape layer from node: " << *node); - shape->setName((layer_info(node) + "_IShapeLayer_for_self").c_str()); - auto shape_out = shape->getOutput(0); - - //output is int[] situation - if (inputs.size() == 1) { - LOG(INFO) << "start converter aten::size(Tensor self) -> (int[])"; - engine->context().set_tensor(node->outputs()[0], shape_out); - LOG(INFO) << "Output tensor shape: " << shape_out->getDimensions(); - //output is int situation - } else { - LOG(INFO) << "start converter aten::size.int(Tensor self, int dim) -> int"; - auto dim = (engine->context().get_constant(inputs[1])).toInt(); - nvinfer1::Dims self_dims = self->getDimensions(); - dim = dim < 0 ? dim + self_dims.nbDims : dim; - - //extract the specific dynamic dim as a 1D-1value tensor - std::vector start_vec{dim}, size_vec{1}, stride_vec{1}; - auto size_layer = engine->network()->addSlice(*shape_out, - sizes_to_nvdim(start_vec), - sizes_to_nvdim(size_vec), - sizes_to_nvdim(stride_vec)); - POROS_CHECK(size_layer, "Unable to given dim info from node: " << *node); - auto size_out = size_layer->getOutput(0); - size_layer->setName((layer_info(node) + "_ISliceLayer_for_size").c_str()); - engine->context().set_tensor(node->outputs()[0], size_out); - LOG(INFO) << "Output tensor shape: " << size_out->getDimensions(); - } - return true; -} - -bool ShapeastensorConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 1), "invaid inputs size for ShapeastensorConverter"); - - //extract self - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - - auto shape = engine->network()->addShape(*self); - POROS_CHECK(shape, "Unable to create shape layer from node: " << *node); - shape->setName((layer_info(node) + "_IShapeLayer_for_self").c_str()); - auto shape_out = shape->getOutput(0); - - engine->context().set_tensor(node->outputs()[0], shape_out); - LOG(INFO) << "Output tensor shape: " << shape_out->getDimensions(); - - return true; -} - -// aten::len.Tensor(Tensor t) -> (int) -// aten::len.t(t[] a) -> (int) -bool LenConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 1), "invaid inputs size for LenConverter"); - - // extract self - auto self = engine->context().get_tensor(inputs[0]); - // POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - if (self != nullptr) { - nvinfer1::Dims self_dims = self->getDimensions(); - if (self_dims.nbDims == 0) { - engine->context().set_constant(node->outputs()[0], 0); - } else if (self_dims.nbDims > 0 && self_dims.d[0] >= 0) { - engine->context().set_constant(node->outputs()[0], self_dims.d[0]); - } else { - // dynamic - nvinfer1::ITensor* self_shape = engine->network()->addShape(*self)->getOutput(0); - self_shape->setName((layer_info(node) + "_IShapeLayer_for_self").c_str()); - - std::vector start_vec{0}, size_vec{1}, stride_vec{1}; - auto slice_layer = engine->network()->addSlice(*self_shape, - sizes_to_nvdim(start_vec), - sizes_to_nvdim(size_vec), - sizes_to_nvdim(stride_vec)); - POROS_CHECK(slice_layer, "Unable to given dim info from node: " << *node); - slice_layer->setName((layer_info(node) + "_ISliceLayer_for_len").c_str()); - auto len_tensor = slice_layer->getOutput(0); - engine->context().set_tensor(node->outputs()[0], len_tensor); - LOG(INFO) << "Output tensor shape: " << len_tensor->getDimensions(); - } - } else { - // tensorlist - if (inputs[0]->type()->isSubtypeOf(c10::ListType::ofTensors())) { - std::vector output_vec; - if (engine->context().get_tensorlist(inputs[0], output_vec)) { - engine->context().set_constant(node->outputs()[0], int(output_vec.size())); - } else { - auto in_const = engine->context().get_constant(inputs[0]); - engine->context().set_constant(node->outputs()[0], int(in_const.toList().size())); - } - // scalarlist - } else if (inputs[0]->type()->isSubtypeOf(c10::ListType::ofInts()) || - inputs[0]->type()->isSubtypeOf(c10::ListType::ofFloats()) || - inputs[0]->type()->isSubtypeOf(c10::ListType::ofBools())) { - auto in_const = engine->context().get_constant(inputs[0]); - engine->context().set_constant(node->outputs()[0], int(in_const.toList().size())); - } else { - POROS_THROW_ERROR("Meet some unsupported output value type in LenConverter" << *node); - } - } - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, AtenSizeConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, ShapeastensorConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, LenConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/shape_handle.h b/poros/poros/converter/gpu/shape_handle.h deleted file mode 100644 index 9bb23ae91b0..00000000000 --- a/poros/poros/converter/gpu/shape_handle.h +++ /dev/null @@ -1,93 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file shape_handle.h -* @author tianjinjin@baidu.com -* @date Mon Nov 29 20:26:44 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class AtenSizeConverter : public GpuConverter { -public: - AtenSizeConverter() {} - virtual ~AtenSizeConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::size(Tensor self) -> (int[])", - "aten::size.int(Tensor self, int dim) -> int"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::size}; - } -}; - -class ShapeastensorConverter : public GpuConverter { -public: - ShapeastensorConverter() {} - virtual ~ShapeastensorConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::_shape_as_tensor(Tensor self) -> (Tensor)"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::_shape_as_tensor}; - } -}; - - -class LenConverter : public GpuConverter { -public: - LenConverter() {} - virtual ~LenConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::len.Tensor(Tensor t) -> (int)", - "aten::len.t(t[] a) -> (int)"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::len}; - } - - bool assign_schema_attr() { - return assign_schema_attr_helper({{"aten::len.t(t[] a) -> (int)", {1, 1}}}); - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/shuffle.cpp b/poros/poros/converter/gpu/shuffle.cpp deleted file mode 100644 index 6d351f8ed01..00000000000 --- a/poros/poros/converter/gpu/shuffle.cpp +++ /dev/null @@ -1,384 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -// Part of the following code in this file refs to -// https://github.com/pytorch/TensorRT/blob/master/core/conversion/converters/impl/shuffle.cpp -// -// Copyright (c) 2020-present, NVIDIA CORPORATION. All rights reserved. -// Copyright (c) Meta Platforms, Inc. and affiliates. -// Licensed under the 3-Clause BSD License - -/** -* @file shuffle.cpp -* @author tianjinjin@baidu.com -* @date Wed Aug 18 16:23:29 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/converter_util.h" -#include "poros/converter/gpu/shuffle.h" -#include "poros/converter/gpu/weight.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/** - * aten::flatten.using_ints(Tensor(a) self, int start_dim=0, int end_dim=-1) -> Tensor(a) - * **/ -bool FlattenConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - //basic check - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 3), "invaid inputs size for FlattenConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for FlattenConverter is not Tensor as expected"); - //assumes int inputs are all come from prim::Constant. - POROS_CHECK_TRUE((inputs[1]->node()->kind() == torch::jit::prim::Constant), - "input[1] for FlattenConverter is not come from prim::Constant as expected"); - POROS_CHECK_TRUE((inputs[2]->node()->kind() == torch::jit::prim::Constant), - "input[2] for FlattenConverter is not come from prim::Constant as expected"); - - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - - auto start_dim = (engine->context().get_constant(inputs[1])).toInt(); - auto end_dim = (engine->context().get_constant(inputs[2])).toInt(); - - auto in_shape = nvdim_to_sizes(in->getDimensions()); - auto in_shape_rank = in_shape.size(); - // 倒序转正序 - start_dim = start_dim < 0 ? start_dim + in_shape_rank : start_dim; - end_dim = end_dim < 0 ? end_dim + in_shape_rank : end_dim; - - POROS_CHECK_TRUE((start_dim >= 0 && (size_t)start_dim < in_shape_rank && - end_dim >= 0 && (size_t)end_dim < in_shape_rank && - start_dim <= end_dim), "invalid start or end dim for node: " << *node); - - std::vector out_shape; - - bool is_dynamic = check_nvtensor_is_dynamic(in); - nvinfer1::IShuffleLayer* shuffle_layer = engine->network()->addShuffle(*in); - POROS_CHECK(shuffle_layer, "Unable to create shuffle layer from node: " << *node); - - if (is_dynamic) { - nvinfer1::ITensor* in_shape = engine->network()->addShape(*in)->getOutput(0); - if (start_dim == end_dim) { - shuffle_layer->setInput(1, *in_shape); - } else { - // Select the dims from start to end with slicelayer and calculate their product. - // Then, concat the result with other dims to get the new shape. - std::vector cat_nvtensor; - - std::vector stride{1}; - std::vector front_start{0}, front_size{start_dim}; - std::vector middle_start{start_dim}, middle_size{end_dim - start_dim + 1}; - std::vector back_start{end_dim + 1}, back_size{(int64_t)in_shape_rank - end_dim - 1}; - - // front - if (start_dim > 0) { - cat_nvtensor.push_back(engine->network()->addSlice(*in_shape, - sizes_to_nvdim(front_start), - sizes_to_nvdim(front_size), - sizes_to_nvdim(stride))->getOutput(0)); - } - // middle - nvinfer1::ITensor* middle_tensor = engine->network()->addSlice(*in_shape, - sizes_to_nvdim(middle_start), - sizes_to_nvdim(middle_size), - sizes_to_nvdim(stride))->getOutput(0); - uint32_t axis_mask = 1; - // axis_mask |= 1 << 1; - nvinfer1::IReduceLayer* reduce_prod_layer = engine->network()->addReduce(*middle_tensor, - nvinfer1::ReduceOperation::kPROD, axis_mask, true); - // default is float32, must set int32 - reduce_prod_layer->setPrecision(nvinfer1::DataType::kINT32); - - cat_nvtensor.push_back(reduce_prod_layer->getOutput(0)); - // back - if ((size_t)end_dim < in_shape_rank - 1) { - cat_nvtensor.push_back(engine->network()->addSlice(*in_shape, - sizes_to_nvdim(back_start), - sizes_to_nvdim(back_size), - sizes_to_nvdim(stride))->getOutput(0)); - } - // cat the new shape - nvinfer1::IConcatenationLayer* concat_layer = - engine->network()->addConcatenation(cat_nvtensor.data(), cat_nvtensor.size()); - concat_layer->setAxis(0); - concat_layer->setName((layer_info(node) + "_IConcatenationLayer").c_str()); - shuffle_layer->setInput(1, *(concat_layer->getOutput(0))); - } - } else { - // static situation - out_shape = torch::flatten(torch::rand(in_shape), start_dim, end_dim).sizes().vec(); - shuffle_layer->setReshapeDimensions(sizes_to_nvdim(out_shape)); - } - - shuffle_layer->setName(layer_info(node).c_str()); - engine->context().set_tensor(node->outputs()[0], shuffle_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << shuffle_layer->getOutput(0)->getDimensions(); - return true; -} - -/** - * aten::permute(Tensor(a) self, int[] dims) -> Tensor(a) - * aten::view(Tensor(a) self, int[] size) -> Tensor(a) - * **/ -bool PermuteViewConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - //basic check - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for PermuteViewConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for PermuteViewConverter is not Tensor as expected"); - //assumes int inputs are all come from prim::Constant. - // POROS_CHECK_TRUE((inputs[1]->node()->kind() == torch::jit::prim::Constant), - // "input[1] for PermuteViewConverter is not come from prim::Constant as expected"); - - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - auto in_shape = nvdim_to_sizes(in->getDimensions()); - - std::vector new_order; - if (!check_inputs_tensor_scalar(engine, node)) { - new_order = (engine->context().get_constant(inputs[1])).toIntList().vec(); - LOG(INFO) << "Shuffle to: " << sizes_to_nvdim(new_order); - } - - auto shuffle = engine->network()->addShuffle(*in); - POROS_CHECK(shuffle, "Unable to create shuffle layer from node: " << *node); - - if (node->kind() == torch::jit::aten::permute) { - nvinfer1::Permutation permute; - std::copy(new_order.begin(), new_order.end(), permute.order); - shuffle->setSecondTranspose(permute); - } else if (node->kind() == torch::jit::aten::view) { - nvinfer1::ITensor* view_size = engine->context().get_tensor(inputs[1]); - if (view_size != nullptr) { - shuffle->setInput(1, *view_size); - } else { - shuffle->setReshapeDimensions(sizes_to_nvdim(new_order)); - } - } else { - POROS_THROW_ERROR("We should never reach here for PermuteViewConverter, meet Unsupported node kind!"); - } - - shuffle->setName(layer_info(node).c_str()); - engine->context().set_tensor(node->outputs()[0], shuffle->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << shuffle->getOutput(0)->getDimensions(); - return true; -} - -/** - * aten::reshape(Tensor(a) self, int[] shape) -> Tensor(a) - * **/ -bool ReshapeConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - //basic check - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for ReshapeConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for ReshapeConverter is not Tensor as expected"); - - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - auto in_shape = nvdim_to_sizes(in->getDimensions()); - - nvinfer1::IShuffleLayer* shuffle_layer = engine->network()->addShuffle(*in); - POROS_CHECK(shuffle_layer, "Unable to create shuffle layer from node: " << *node); - - // 检查是否能使用get_tensor获取input[1] - if (engine->context().get_tensor(inputs[1]) != nullptr) { - nvinfer1::ITensor* new_shape = engine->context().get_tensor(inputs[1]); - shuffle_layer->setInput(1, *new_shape); - } else { - std::vector new_order = (engine->context().get_constant(inputs[1])).toIntList().vec(); - // if input shape is dynamic, torch::reshape is wrong. - // std::vector new_shape = torch::reshape(torch::rand(in_shape), new_order).sizes().vec(); - LOG(INFO) << "Shuffle to: " << sizes_to_nvdim(new_order); - shuffle_layer->setReshapeDimensions(sizes_to_nvdim(new_order)); - } - - shuffle_layer->setName(layer_info(node).c_str()); - engine->context().set_tensor(node->outputs()[0], shuffle_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << shuffle_layer->getOutput(0)->getDimensions(); - return true; -} - -/** - * aten::transpose.int(Tensor(a) self, int dim0, int dim1) -> Tensor(a) - * **/ -bool TransposeConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - //basic check - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 3), "invaid inputs size for TransposeConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for TransposeConverter is not Tensor as expected"); - //assumes int inputs are all come from prim::Constant. - POROS_CHECK_TRUE((inputs[1]->node()->kind() == torch::jit::prim::Constant), - "input[1] for TransposeConverter is not come from prim::Constant as expected"); - POROS_CHECK_TRUE((inputs[2]->node()->kind() == torch::jit::prim::Constant), - "input[2] for TransposeConverter is not come from prim::Constant as expected"); - - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - auto in_shape = nvdim_to_sizes(in->getDimensions()); - auto ndims = in_shape.size(); - - //extract dim0 & dim1 - auto dim0 = (engine->context().get_constant(inputs[1])).toInt(); - auto dim1 = (engine->context().get_constant(inputs[2])).toInt(); - - std::vector new_order; - for (size_t i = 0; i < ndims; i++) { - new_order.push_back(i); - } - dim0 = dim0 < 0 ? (dim0 + ndims) : dim0; - dim1 = dim1 < 0 ? (dim1 + ndims) : dim1; - auto tmp = dim0; - new_order[dim0] = new_order[dim1]; - new_order[dim1] = tmp; - LOG(INFO) << "Shuffle to: " << sizes_to_nvdim(new_order); - - auto shuffle = engine->network()->addShuffle(*in); - POROS_CHECK(shuffle, "Unable to create shuffle layer from node: " << *node); - nvinfer1::Permutation permute; - std::copy(new_order.begin(), new_order.end(), permute.order); - shuffle->setSecondTranspose(permute); - shuffle->setName(layer_info(node).c_str()); - engine->context().set_tensor(node->outputs()[0], shuffle->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << shuffle->getOutput(0)->getDimensions(); - return true; -} - -/** - * aten::t(Tensor(a) self) -> Tensor(a) - * **/ -bool AtenTConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - //basic check - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 1), "invaid inputs size for AtenTConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for AtenTConverter is not Tensor as expected"); - - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - auto input_dims = in->getDimensions(); - - if (input_dims.nbDims < 2) { - //For aten::t situation. if input tensors < 2D, return them as is - engine->context().set_tensor(node->outputs()[0], in); - LOG(INFO) << "Output tensor shape: " << in->getDimensions(); - return true; - } - - auto shuffle = engine->network()->addShuffle(*in); - POROS_CHECK(shuffle, "Unable to create shuffle layer from node: " << *node); - nvinfer1::Permutation first_perm; - first_perm.order[0] = 1; - first_perm.order[1] = 0; - shuffle->setFirstTranspose(first_perm); - shuffle->setZeroIsPlaceholder(false); - shuffle->setName(layer_info(node).c_str()); - engine->context().set_tensor(node->outputs()[0], shuffle->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << shuffle->getOutput(0)->getDimensions(); - return true; -} - -/** - * aten::pixel_shuffle(Tensor self, int upscale_factor) -> Tensor - * **/ -bool PixelShuffleConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - //basic check - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for PixelShuffleConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for PixelShuffleConverter is not Tensor as expected"); - //assumes int inputs are all come from prim::Constant. - POROS_CHECK_TRUE((inputs[1]->node()->kind() == torch::jit::prim::Constant), - "input[1] for PixelShuffleConverter is not come from prim::Constant as expected"); - - //extract self - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - auto in_shape = nvdim_to_sizes(self->getDimensions()); - int64_t irank = in_shape.size(); - POROS_CHECK(irank >= 3, "pixel_shuffle expects input to have at least 3 dimensions, but got input with " - << std::to_string(irank) << " dimension(s)"); - - //extract upscale_factor - int64_t upscale_factor = (engine->context().get_constant(inputs[1])).toInt(); - POROS_CHECK(upscale_factor > 0, "pixel_shuffle expects a positive upscale_factor, but got " - << std::to_string(upscale_factor)); - int64_t upscale_factor_squared = upscale_factor * upscale_factor; - - - const auto NUM_NON_BATCH_DIMS = 3; - const auto self_sizes_batch_end = in_shape.end() - NUM_NON_BATCH_DIMS; - - int64_t ic = in_shape[irank - 3]; - int64_t ih = in_shape[irank - 2]; - int64_t iw = in_shape[irank - 1]; - POROS_CHECK(ic % upscale_factor_squared == 0, - "pixel_shuffle expects its input's 'channel' dimension to be divisible by the square of " - << "upscale_factor, but input.size(-3)=" << std::to_string(ic) << " is not divisible by " - << std::to_string(upscale_factor_squared)); - - int64_t oc = ic / upscale_factor_squared; - int64_t oh = ih * upscale_factor; - int64_t ow = iw * upscale_factor; - - std::vector added_dims_shape(in_shape.begin(), self_sizes_batch_end); - added_dims_shape.insert(added_dims_shape.end(), {oc, upscale_factor, upscale_factor, ih, iw}); - auto view_layer = engine->network()->addShuffle(*self); - POROS_CHECK(view_layer, "Unable to create shuffle layer from node: " << *node); - view_layer->setReshapeDimensions(sizes_to_nvdim(added_dims_shape)); - int64_t view_rank = added_dims_shape.size(); - - auto permutation_layer = engine->network()->addShuffle(*view_layer->getOutput(0)); - POROS_CHECK(permutation_layer, "Unable to create shuffle layer from node: " << *node); - std::vector new_order(in_shape.begin(), self_sizes_batch_end); - std::iota(new_order.begin(), new_order.end(), 0); - new_order.insert( - new_order.end(), - {view_rank - 5, view_rank - 2, view_rank - 4, view_rank - 1, view_rank - 3}); - nvinfer1::Permutation permute; - std::copy(new_order.begin(), new_order.end(), permute.order); - permutation_layer->setSecondTranspose(permute); - - - std::vector final_shape(in_shape.begin(), self_sizes_batch_end); - final_shape.insert(final_shape.end(), {oc, oh, ow}); - auto last_view_layer = engine->network()->addShuffle(*permutation_layer->getOutput(0)); - POROS_CHECK(last_view_layer, "Unable to create shuffle layer from node: " << *node); - last_view_layer->setReshapeDimensions(sizes_to_nvdim(final_shape)); - last_view_layer->setName(layer_info(node).c_str()); - engine->context().set_tensor(node->outputs()[0], last_view_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << last_view_layer->getOutput(0)->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, FlattenConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, PermuteViewConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, ReshapeConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, TransposeConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, AtenTConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, PixelShuffleConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/shuffle.h b/poros/poros/converter/gpu/shuffle.h deleted file mode 100644 index 81bac43c0c0..00000000000 --- a/poros/poros/converter/gpu/shuffle.h +++ /dev/null @@ -1,163 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file shuffle.h -* @author tianjinjin@baidu.com -* @date Wed Aug 18 15:37:48 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class FlattenConverter : public GpuConverter { -public: - FlattenConverter() {} - virtual ~FlattenConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::flatten.using_ints(Tensor(a) self, int start_dim=0, int end_dim=-1) -> Tensor(a)"}; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * aten::flatten.named_out_dim(Tensor(a) self, int start_dim, int end_dim, Dimname out_dim) -> Tensor(a) - * aten::flatten.using_names(Tensor(a) self, Dimname start_dim, Dimname end_dim, Dimname out_dim) -> Tensor(a) - * aten::flatten.DimnameList(Tensor(a) self, Dimname[] dims, Dimname out_dim) -> Tensor(a) - * **/ - - const std::vector node_kind() { - return {torch::jit::aten::flatten}; - } -}; - - -class PermuteViewConverter : public GpuConverter { -public: - PermuteViewConverter() {} - virtual ~PermuteViewConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::permute(Tensor(a) self, int[] dims) -> Tensor(a)", - "aten::view(Tensor(a) self, int[] size) -> Tensor(a)"}; - } - - /** TODO: TRY TO SUPPORT SCHEMA PATTERNS BELLOW: - * aten::view.dtype(Tensor(a) self, ScalarType dtype) -> Tensor(a) - **/ - const std::vector node_kind() { - return {torch::jit::aten::permute, - torch::jit::aten::view}; - } - - bool assign_schema_attr() { - return assign_schema_attr_helper({{"aten::view(Tensor(a) self, int[] size) -> Tensor(a)", {1, 1}}}); - } -}; - -class ReshapeConverter : public GpuConverter { -public: - ReshapeConverter() {} - virtual ~ReshapeConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::reshape(Tensor(a) self, int[] shape) -> Tensor(a)"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::reshape}; - } - - bool assign_schema_attr() { - return assign_schema_attr_helper({{"aten::reshape(Tensor(a) self, int[] shape) -> Tensor(a)", {1, 1}}}); - } -}; - -class TransposeConverter : public GpuConverter { -public: - TransposeConverter() {} - virtual ~TransposeConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::transpose.int(Tensor(a) self, int dim0, int dim1) -> Tensor(a)"}; - } - - /** TODO: TRY TO SUPPORT SCHEMA PATTERNS BELLOW: - * aten::transpose.Dimname(Tensor(a) self, Dimname dim0, Dimname dim1) -> Tensor(a) - * aten::transpose_(Tensor(a!) self, int dim0, int dim1) -> Tensor(a!) - **/ - const std::vector node_kind() { - return {torch::jit::aten::transpose}; - } -}; - -class AtenTConverter : public GpuConverter { -public: - AtenTConverter() {} - virtual ~AtenTConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::t(Tensor(a) self) -> Tensor(a)"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::t}; - } -}; - -class PixelShuffleConverter : public GpuConverter { -public: - PixelShuffleConverter() {} - virtual ~PixelShuffleConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::pixel_shuffle(Tensor self, int upscale_factor) -> Tensor"}; - } - - const std::vector node_kind() { - return {torch::jit::aten::pixel_shuffle}; - } - - bool assign_schema_attr() { - return assign_schema_attr_helper({{"aten::pixel_shuffle(Tensor self, int upscale_factor) -> Tensor", {0, 0}}}); - } -}; - - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/converter/gpu/softmax.cpp b/poros/poros/converter/gpu/softmax.cpp deleted file mode 100644 index 0b1857e1848..00000000000 --- a/poros/poros/converter/gpu/softmax.cpp +++ /dev/null @@ -1,118 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file softmax.cpp -* @author tianjinjin@baidu.com -* @date Tue Aug 24 17:15:33 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/softmax.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/*aten::softmax.int(Tensor self, int dim, ScalarType? dtype=None) -> Tensor*/ -bool SoftmaxConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 3), "invaid inputs size for SoftmaxConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for SoftmaxConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[1]->node()->kind() == torch::jit::prim::Constant), - "input[1] for SoftmaxConverter is not come from prim::Constant as expected"); - LOG(INFO) << "Disregarding input[2] dtype argument"; - - auto in = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((in != nullptr), "Unable to init input tensor for node: " << *node); - auto shape = nvdim_to_sizes(in->getDimensions()); - - bool is_dynamic = check_nvtensor_is_dynamic(in); - nvinfer1::ITensor* in_shape_tensor = nullptr; - if (is_dynamic) { - in_shape_tensor = engine->network()->addShape(*in)->getOutput(0); - } - // SoftMax needs at least 2D input - if (shape.size() < 2) { - auto new_shape = sizes_to_nvdim_with_pad(shape, 2); - auto shuffle = engine->network()->addShuffle(*in); - shuffle->setReshapeDimensions(new_shape); - shuffle->setName((layer_info(node) + " [Reshape to " + nvdim_to_str(new_shape) + ']').c_str()); - if (is_dynamic) { - nvinfer1::ITensor* insert_tensor = tensor_to_const(engine, torch::tensor({1}, torch::kInt32)); - std::vector inputs_nvtensor; - inputs_nvtensor.push_back(insert_tensor); - inputs_nvtensor.push_back(in_shape_tensor); - nvinfer1::IConcatenationLayer* concat_layer = - engine->network()->addConcatenation(inputs_nvtensor.data(), inputs_nvtensor.size()); - concat_layer->setAxis(0); - concat_layer->setName((layer_info(node) + "_IConcatenationLayer").c_str()); - nvinfer1::ITensor* concat_out = concat_layer->getOutput(0); - shuffle->setInput(1, *concat_out); - shuffle->setName((layer_info(node) + "_IShuffleLayer_1D_to_2D").c_str()); - } - in = shuffle->getOutput(0); - } - - //extract dim - auto dim = (engine->context().get_constant(inputs[1])).toInt(); - if (dim < 0) { - dim = shape.size() + dim; - } - - //main function - auto softmax = engine->network()->addSoftMax(*in); - POROS_CHECK(softmax, "Unable to create softmax layer from node: " << *node); - if (shape.size() > 1) { - softmax->setAxes(1 << (dim)); - } else { - // When there is no batch dimension - softmax->setAxes(1 << (dim + 1)); - } - softmax->setName((layer_info(node) + "_ISoftMaxLayer").c_str()); - auto out_tensor = softmax->getOutput(0); - - // SoftMax reshape back - if (shape.size() < 2) { - auto old_shape = sizes_to_nvdim(shape); - LOG(INFO) << "Input shape was less than 2D got: " << old_shape - << ", inserting shuffle layer to reshape back"; - auto shuffle = engine->network()->addShuffle(*out_tensor); - shuffle->setReshapeDimensions(old_shape); - shuffle->setName((layer_info(node) + " [Reshape to " + nvdim_to_str(old_shape) + ']').c_str()); - if (is_dynamic) { - shuffle->setInput(1, *in_shape_tensor); - shuffle->setName((layer_info(node) + "shuffle_to_old_shape").c_str()); - } - out_tensor = shuffle->getOutput(0); - } - - engine->context().set_tensor(node->outputs()[0], out_tensor); - LOG(INFO) << "Output tensor shape: " << out_tensor->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, SoftmaxConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/softmax.h b/poros/poros/converter/gpu/softmax.h deleted file mode 100644 index 9e0fdaf5ab3..00000000000 --- a/poros/poros/converter/gpu/softmax.h +++ /dev/null @@ -1,57 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file softmax.h -* @author tianjinjin@baidu.com -* @date Tue Aug 24 17:15:33 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class SoftmaxConverter : public GpuConverter { -public: - SoftmaxConverter() {} - virtual ~SoftmaxConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::softmax.int(Tensor self, int dim, ScalarType? dtype=None) -> Tensor"}; - } - - /** TODO: TRY TO SUPPORT SCHEMA PATTERNS BELLOW: - * aten::softmax.Dimname(Tensor self, Dimname dim, *, ScalarType? dtype=None) -> Tensor - **/ - const std::vector node_kind() { - return {torch::jit::aten::softmax}; - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/squeeze.cpp b/poros/poros/converter/gpu/squeeze.cpp deleted file mode 100644 index 77aed228545..00000000000 --- a/poros/poros/converter/gpu/squeeze.cpp +++ /dev/null @@ -1,206 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file squeeze.cpp -* @author tianjinjin@baidu.com -* @date Wed Sep 1 11:19:13 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/squeeze.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -nvinfer1::IShuffleLayer* add_shuffle_layer(TensorrtEngine* engine, const torch::jit::Node *node, \ - nvinfer1::ITensor* input, int64_t dim, int64_t idx) { - auto shuffle_layer = engine->network()->addShuffle(*input); - POROS_CHECK(shuffle_layer, "Unable to create shuffle layer from node: " << *node); - shuffle_layer->setName((layer_info(node) + "_IShuffleLayer_for_index_" + std::to_string(idx)).c_str()); - nvinfer1::ITensor* input_shape_tensor = (engine->network()->addShape(*input))->getOutput(0); - nvinfer1::ITensor* reshape_tensor = squeeze_nv_shapetensor(engine, input_shape_tensor, dim); - - if (reshape_tensor != nullptr) { - shuffle_layer->setInput(1, *reshape_tensor); - } else { - LOG(INFO) << "squeeze nv shape tensor error!"; - return nullptr; - } - return shuffle_layer; -} - - -/* -"aten::squeeze.dim(Tensor(a) self, int dim) -> Tensor(a)", -https://pytorch.org/docs/stable/generated/torch.squeeze.html -将输入张量形状中的1去除并返回。 如果输入是形如(A×1×B×1×C×1×D),那么输出形状就为: (A×B×C×D)*/ -bool SqueezeConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2 || inputs.size() == 1), "invaid inputs size for SqueezeConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for SqueezeConverter is not Tensor as expected"); - if (inputs.size() == 2) { - POROS_CHECK_TRUE((inputs[1]->node()->kind() == torch::jit::prim::Constant), - "input[1] for SqueezeConverter is not come from prim::Constant as expected"); - } - - //extract self - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - - std::vector dims; - int64_t sign = 0; - if (inputs.size() == 1) { - // 这里目前只支持非dynamic - auto shape = self->getDimensions().d; - for (int i = 0; i < self->getDimensions().nbDims; i++) { - if (shape[i] == 1) { - dims.push_back(i - sign); - sign += 1; - } - } - if (dims.size() == 0) { - return true; - } - } - else { - //extract dim - auto dim = (engine->context().get_constant(inputs[1])).toInt(); - auto self_dim = nvdim_to_sizes(self->getDimensions()); - if (dim < 0) { - dim = self_dim.size() + dim; - } - - if (self_dim[dim] != 1) { - //不需要squeeze的情况 - engine->context().set_tensor(node->outputs()[0], self); - LOG(INFO) << "Output tensor shape: " << self->getDimensions(); - return true; - } else { - dims = {dim}; - } - } - - bool is_dynamic = check_nvtensor_is_dynamic(self); - nvinfer1::IShuffleLayer* shuffle_layer = nullptr; - if (is_dynamic) { - shuffle_layer = add_shuffle_layer(engine, node, self, dims[0], 0); - POROS_CHECK(shuffle_layer, "Unable to create shuffle layer from node: " << *node) - if (nullptr == shuffle_layer){ - LOG(INFO) << "unsqueeze nv shape tensor error!"; - return false; - } - for (size_t i = 1; i < dims.size(); i++) { - shuffle_layer = add_shuffle_layer(engine, node, shuffle_layer->getOutput(0), dims[i], i); - POROS_CHECK(shuffle_layer, "Unable to create shuffle layer from node: " << *node) - } - engine->context().set_tensor(node->outputs()[0], shuffle_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << shuffle_layer->getOutput(0)->getDimensions(); - } else { - shuffle_layer = engine->network()->addShuffle(*self); - POROS_CHECK(shuffle_layer, "Unable to create shuffle layer from node: " << *node); - shuffle_layer->setName((layer_info(node) + "_IShuffleLayer_for_self").c_str()); - for (size_t i = 0; i < dims.size(); i++) { - if (i == 0) { - shuffle_layer->setReshapeDimensions(squeeze_dims(self->getDimensions(), dims[i])); - } else { - shuffle_layer->setReshapeDimensions(squeeze_dims(shuffle_layer->getOutput(0)->getDimensions(), dims[i])); - } - - if (i != dims.size() - 1) { - shuffle_layer = engine->network()->addShuffle(*shuffle_layer->getOutput(0)); - shuffle_layer->setName((layer_info(node) + "_IShuffleLayer_for_output").c_str()); - } - } - } - engine->context().set_tensor(node->outputs()[0], shuffle_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << shuffle_layer->getOutput(0)->getDimensions(); - return true; -} - -/* -"aten::unsqueeze(Tensor(a) self, int dim) -> Tensor(a)", -https://pytorch.org/docs/stable/generated/torch.unsqueeze.html*/ -bool UnSqueezeConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 2), "invaid inputs size for UnSqueezeConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for UnSqueezeConverter is not Tensor as expected"); - POROS_CHECK_TRUE((inputs[1]->node()->kind() == torch::jit::prim::Constant), - "input[1] for UnSqueezeConverter is not come from prim::Constant as expected"); - - //extract self - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - - //extract dim - auto dim = (engine->context().get_constant(inputs[1])).toInt(); - if (self->getDimensions().nbDims == 0 && dim == 0) { - auto shuffle_layer = engine->network()->addShuffle(*self); - nvinfer1::Dims unsqueeze_dim; - unsqueeze_dim.nbDims = 1; - unsqueeze_dim.d[0] = 1; - shuffle_layer->setReshapeDimensions(unsqueeze_dim); - shuffle_layer->setName((layer_info(node) + "_IShuffleLayer_for_self").c_str()); - auto output = shuffle_layer->getOutput(0); - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - return true; - } - auto self_dim = nvdim_to_sizes(self->getDimensions()); - int64_t nbDims = self_dim.size(); - POROS_CHECK((dim <= nbDims && dim >= -(nbDims + 1)), - "Dimension out of range (expected to be in range of [" << -(nbDims + 1) - << ", " << nbDims << "], but got " << dim << ")"); - if (dim < 0) { - dim = self_dim.size() + dim + 1; - } - - auto shuffle_layer = engine->network()->addShuffle(*self); - POROS_CHECK(shuffle_layer, "Unable to create shuffle layer from node: " << *node); - bool is_dynamic = check_nvtensor_is_dynamic(self); - if (is_dynamic) { - nvinfer1::ITensor* input_shape_tensor = (engine->network()->addShape(*self))->getOutput(0); - nvinfer1::ITensor* reshape_tensor = unsqueeze_nv_shapetensor(engine, input_shape_tensor, dim); - if (reshape_tensor != nullptr) { - shuffle_layer->setInput(1, *reshape_tensor); - } else { - LOG(INFO) << "unsqueeze nv shape tensor error!"; - return false; - } - } else { - shuffle_layer->setReshapeDimensions(unsqueeze_dims(self->getDimensions(), dim)); - } - shuffle_layer->setName((layer_info(node) + "_IShuffleLayer_for_self").c_str()); - engine->context().set_tensor(node->outputs()[0], shuffle_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << shuffle_layer->getOutput(0)->getDimensions(); - - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, SqueezeConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, UnSqueezeConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/squeeze.h b/poros/poros/converter/gpu/squeeze.h deleted file mode 100644 index aca128578a5..00000000000 --- a/poros/poros/converter/gpu/squeeze.h +++ /dev/null @@ -1,76 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file squeeze.h -* @author tianjinjin@baidu.com -* @date Wed Sep 1 11:19:13 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class SqueezeConverter : public GpuConverter { -public: - SqueezeConverter() {} - virtual ~SqueezeConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::squeeze.dim(Tensor(a) self, int dim) -> Tensor(a)", - "aten::squeeze(Tensor(a) self) -> (Tensor(a))"}; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::squeeze(Tensor(a) self) -> Tensor(a)", - * "aten::squeeze.dimname(Tensor(a) self, Dimname dim) -> Tensor(a)" - * **/ - const std::vector node_kind() { - return {torch::jit::aten::squeeze}; - } -}; - -class UnSqueezeConverter : public GpuConverter { -public: - UnSqueezeConverter() {} - virtual ~UnSqueezeConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::unsqueeze(Tensor(a) self, int dim) -> Tensor(a)", - }; - } - - const std::vector node_kind() { - return {torch::jit::aten::unsqueeze}; - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/stack.cpp b/poros/poros/converter/gpu/stack.cpp deleted file mode 100644 index 8cd3a315bd3..00000000000 --- a/poros/poros/converter/gpu/stack.cpp +++ /dev/null @@ -1,110 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file stack.cpp -* @author tianjinjin@baidu.com -* @date Tue Sep 7 15:09:14 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/stack.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/* -"aten::stack(Tensor[] tensors, int dim=0) -> Tensor", -"aten::vstack(Tensor[] tensors) -> Tensor" -*/ - -bool StackConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 1 || inputs.size() == 2), "invaid inputs size for StackConverter"); - POROS_CHECK_TRUE(inputs[0]->type()->isSubtypeOf(c10::ListType::ofTensors()), - "input[0] for StackConverter is not TensorList as expected"); - - //extract tensors - std::vector tensorlist; - POROS_CHECK_TRUE((engine->context().get_tensorlist(inputs[0], tensorlist)), "extract tensorlist error"); - - int64_t dim = 0; - - std::vector tensors; - if (inputs.size() == 2) { - // aten::stack - POROS_CHECK_TRUE(inputs[1]->type()->isSubtypeOf(c10::NumberType::get()), - "input[1] for StackConverter is not int64_t as expected"); - - //extract dims - dim = (engine->context().get_constant(inputs[1])).toInt(); - - // aten::stack should unsqueeze dims - // check if input tensorlist is dynamic. - bool is_dynamic = check_nvtensor_is_dynamic(tensorlist[0]); - nvinfer1::Dims inputs_dims = tensorlist[0]->getDimensions(); - - // when dim is negtive - if (dim < 0) { - dim = inputs_dims.nbDims + dim + 1; - } - // generate unsqueeze dimensions by shapetensor if dynamic. - nvinfer1::ITensor* unsqueeze_dim = nullptr; - if (is_dynamic) { - nvinfer1::ITensor* input_shapetensor = engine->network()->addShape(*(tensorlist[0]))->getOutput(0); - unsqueeze_dim = unsqueeze_nv_shapetensor(engine, input_shapetensor, dim); - if (unsqueeze_dim == nullptr) { - LOG(INFO) << "unsqueeze nv shape tensor failed"; - return false; - } - } - // unsqueeze each tensor in tensorlist - for (size_t i = 0; i < tensorlist.size(); ++i) { - auto shuffle_layer = engine->network()->addShuffle(*tensorlist[i]); - POROS_CHECK(shuffle_layer, "Unable to create shuffle layer from node: " << *node); - if (is_dynamic) { - shuffle_layer->setInput(1, *unsqueeze_dim); - } else { - shuffle_layer->setReshapeDimensions(unsqueeze_dims(tensorlist[i]->getDimensions(), dim)); - } - shuffle_layer->setName((layer_info(node) + "_IShuffleLayer_for_tensor_" + std::to_string(i)).c_str()); - tensors.push_back(shuffle_layer->getOutput(0)); - } - } else { - // aten::vstack need not unsqueeze dims - tensors = tensorlist; - } - - auto cat_layer = engine->network()->addConcatenation(tensors.data(), tensors.size()); - POROS_CHECK(cat_layer, "Unable to create concatenation layer from node: " << *node); - cat_layer->setAxis(static_cast(dim)); - cat_layer->setName((layer_info(node) + "_IConcatenationLayer").c_str()); - engine->context().set_tensor(node->outputs()[0], cat_layer->getOutput(0)); - LOG(INFO) << "Output tensor shape: " << cat_layer->getOutput(0)->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, StackConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/stack.h b/poros/poros/converter/gpu/stack.h deleted file mode 100644 index e400e3fd16f..00000000000 --- a/poros/poros/converter/gpu/stack.h +++ /dev/null @@ -1,65 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file stack.h -* @author tianjinjin@baidu.com -* @date Tue Sep 7 15:09:14 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class StackConverter : public GpuConverter { -public: - StackConverter() {} - virtual ~StackConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::stack(Tensor[] tensors, int dim=0) -> Tensor", - "aten::vstack(Tensor[] tensors) -> Tensor" - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::stack.out(Tensor[] tensors, int dim=0, *, Tensor(a!) out) -> Tensor(a!)", - * **/ - const std::vector node_kind() { - return {torch::jit::aten::stack, - torch::jit::aten::vstack, - }; - } - - bool assign_schema_attr() { - return assign_schema_attr_helper({{"aten::stack(Tensor[] tensors, int dim=0) -> Tensor", {1, 1}}}); - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/to.cpp b/poros/poros/converter/gpu/to.cpp deleted file mode 100644 index 32373f408bb..00000000000 --- a/poros/poros/converter/gpu/to.cpp +++ /dev/null @@ -1,140 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file to.cpp -* @author wangrui39@baidu.com -* @date Saturday November 13 11:36:11 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/to.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/engine_context.h" -#include "poros/util/macros.h" -#include "poros/context/poros_global.h" -#include "poros/converter/gpu/weight.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -static void long_to_int(at::ScalarType &scalar_type) { - if (scalar_type == at::kLong && PorosGlobalContext::instance().get_poros_options().long_to_int == true) { - scalar_type = at::kInt; - LOG(WARNING) << "gen_tensor_type meets at::KLong tensor type, change this to at::KInt. " - << "Attention: this may leed to percision change"; - } -} - -bool ToConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 5 || inputs.size() == 6 || - inputs.size() == 8), "invaid inputs size for ToConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "inputs[0] for ToConverter is not Tensor as expected"); - auto self = engine->context().get_tensor(inputs[0]); - nvinfer1::DataType output_type = self->getType(); - - // aten::to.other(Tensor self, Tensor other, bool non_blocking=False, bool copy=False, MemoryFormat? memory_format=None) -> Tensor - if (inputs[1]->type()->isSubtypeOf(c10::TensorType::get())) { - auto other = engine->context().get_tensor(inputs[1]); - output_type = other->getType(); - - // aten::to.device(Tensor self, Device device, ScalarType dtype, bool non_blocking=False, bool copy=False, MemoryFormat? memory_format=None) -> Tensor - // aten::to.prim_Device(Tensor(a) self, Device? device, int? dtype=None, bool non_blocking=False, bool copy=False) -> (Tensor(b|a)) - } else if (inputs[2]->type()->str() == "int" && inputs[3]->type()->str() == "bool") { - auto scalar_type = engine->context().get_constant(inputs[2]).toScalarType(); - long_to_int(scalar_type); - output_type = attype_to_nvtype(scalar_type); - if (engine->context().get_constant(inputs[1]).toDevice().is_cuda()) { - auto device = nvinfer1::TensorLocation::kDEVICE; - self->setLocation(device); - } else { - LOG(WARNING) << "Set tensor device to HOST but only DEVICE is supported"; - return false; - } - - /* aten::to.dtype(Tensor self, ScalarType dtype, bool non_blocking=False, bool copy=False, MemoryFormat? memory_format=None) -> Tensor - aten::to.dtype_layout(Tensor self, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None, - bool non_blocking=False, bool copy=False, int? memory_format=None) -> (Tensor)*/ - } else if (inputs[1]->type()->str() == "int") { - auto scalar_type = engine->context().get_constant(inputs[1]).toScalarType(); - long_to_int(scalar_type); - output_type = attype_to_nvtype(scalar_type); - // Input err - } else { - POROS_THROW_ERROR("Meet some unsupported inputs value type in ToConstructConverter" << *node); - return false; - } - - // Set datatype for self to dtype - // 注:尽管此处output_type可能和input_type一样,但保险起见也需要过一下identity_layer,否则execute_engine时可能发生错误。 - // 例如:aten::to的input和output tensor同时被mark成engine的输出,如果不走identity_layer,那么这两个tensor其实是一个tensor。 - // build_engine时trt会报 xxx has been marked as output(trt不支持重复标记输出,只会覆盖之前的输出。) - // 原本期望输出两个实际却只有一个,execute_engine获取输出时会出core。 - // todo: 同类型转换的aten::to.dtype也可以在graph中干掉 - auto identity = engine->network()->addIdentity(*self); - identity->setOutputType(0, output_type); - identity->setName((layer_info(node) + "_IIdentityLayer_for_self").c_str()); - self = identity->getOutput(0); - // setOutputType可能不起作用,用setType再次确保self的类型发生了转换 - self->setType(output_type); - - engine->context().set_tensor(node->outputs()[0], self); - LOG(INFO) << "Output tensor shape: " << self->getDimensions(); - return true; -} - -// prim::NumToTensor.Scalar(Scalar a) -> (Tensor) -bool NumtotensorConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 1), "invaid inputs size for NumtotensorConverter"); - // 如果是tensor封装的scalar直接向后传 - nvinfer1::ITensor* self = engine->context().get_tensor(inputs[0]); - if (self != nullptr) { - engine->context().set_tensor(node->outputs()[0], self); - LOG(INFO) << "Output tensor shape: " << self->getDimensions(); - } else { - // 如果传入的是真实的scalar - auto input_scalar = engine->context().get_constant(inputs[0]); - if (!input_scalar.isScalar()) { - POROS_THROW_ERROR("prim::NumToTensor input[0] is not scalar!"); - return false; - } - nvinfer1::ITensor* output_tensor = nullptr; - if (input_scalar.isInt()) { - output_tensor = tensor_to_const(engine, at::tensor(input_scalar.toInt(), torch::kInt)); - } else if (input_scalar.isDouble()) { - output_tensor = tensor_to_const(engine, at::tensor(input_scalar.toDouble(), torch::kDouble).to(at::ScalarType::Float)); - } else if (input_scalar.isBool()) { - output_tensor = tensor_to_const(engine, at::tensor(input_scalar.toBool(), torch::kBool)); - } else { - POROS_THROW_ERROR("prim::NumToTensor Converter meets an unsupported scalar type, which leads to fail."); - return false; - } - engine->context().set_tensor(node->outputs()[0], output_tensor); - LOG(INFO) << "Output tensor shape: " << output_tensor->getDimensions(); - } - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, ToConverter); -POROS_REGISTER_CONVERTER(TensorrtEngine, NumtotensorConverter); - -} // baidu -} // mirana -} // poros diff --git a/poros/poros/converter/gpu/to.h b/poros/poros/converter/gpu/to.h deleted file mode 100644 index 83f300f7416..00000000000 --- a/poros/poros/converter/gpu/to.h +++ /dev/null @@ -1,82 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file to.h -* @author wangrui39@baidu.com -* @date Saturday November 13 11:36:11 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -// Correspons to torch.tensor.to https://pytorch.org/docs/1.9.0/generated/torch.Tensor.to.html?highlight=#torch.to -class ToConverter : public GpuConverter { -public: - ToConverter() {} - virtual ~ToConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::to.device(Tensor self, Device device, ScalarType dtype, bool non_blocking=False, bool copy=False, MemoryFormat? memory_format=None) -> Tensor", - "aten::to.dtype(Tensor self, ScalarType dtype, bool non_blocking=False, bool copy=False, MemoryFormat? memory_format=None) -> Tensor", - "aten::to.other(Tensor self, Tensor other, bool non_blocking=False, bool copy=False, MemoryFormat? memory_format=None) -> Tensor", - "aten::to.dtype_layout(Tensor self, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None, bool non_blocking=False, bool copy=False, int? memory_format=None) -> (Tensor)", - "aten::to.prim_Device(Tensor(a) self, Device? device, int? dtype=None, bool non_blocking=False, bool copy=False) -> (Tensor(b|a))", - }; - } - - const std::vector node_kind() { - return {torch::jit::aten::to}; - } -}; - -class NumtotensorConverter : public GpuConverter { -public: - NumtotensorConverter() {} - virtual ~NumtotensorConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"prim::NumToTensor.Scalar(Scalar a) -> (Tensor)", - }; - } - - const std::vector node_kind() { - return {torch::jit::prim::NumToTensor}; - } - - bool assign_schema_attr() { - return assign_schema_attr_helper({{"prim::NumToTensor.Scalar(Scalar a) -> (Tensor)", {1, 1}}}); - } -}; - - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/converter/gpu/topk.cpp b/poros/poros/converter/gpu/topk.cpp deleted file mode 100644 index e43bba11bb8..00000000000 --- a/poros/poros/converter/gpu/topk.cpp +++ /dev/null @@ -1,78 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file topk.cpp -* @author tianjinjin@baidu.com -* @date Tue Sep 7 14:29:20 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/topk.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/* -"aten::topk(Tensor self, -int k, -int dim=-1, -bool largest=True, -bool sorted=True) -> (Tensor values, Tensor indices)", -*/ -bool TopkConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 5), "invaid inputs size for TopkConverter"); - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for TopkConverter is not Tensor as expected"); - - //extract self - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - auto self_dim = nvdim_to_sizes(self->getDimensions()); - - //extract k & dim & largest - auto k = (engine->context().get_constant(inputs[1])).toInt(); - auto dim = (engine->context().get_constant(inputs[2])).toInt(); - auto largest = (engine->context().get_constant(inputs[3])).toBool(); - - if (dim < 0) { - dim = self_dim.size() + dim; - } - uint32_t shift_dim = 1 << dim; - auto topk_type = largest ? (nvinfer1::TopKOperation::kMAX) : (nvinfer1::TopKOperation::kMIN); - auto new_layer = engine->network()->addTopK(*self, topk_type, k, shift_dim); - - POROS_CHECK(new_layer, "Unable to create topk layer from node: " << *node); - new_layer->setName((layer_info(node) + "_ITopKLayer").c_str()); - engine->context().set_tensor(node->outputs()[0], new_layer->getOutput(0)); - engine->context().set_tensor(node->outputs()[1], new_layer->getOutput(1)); - LOG(INFO) << "Output tensor(0) shape: " << new_layer->getOutput(0)->getDimensions(); - LOG(INFO) << "Output tensor(1) shape: " << new_layer->getOutput(1)->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, TopkConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/topk.h b/poros/poros/converter/gpu/topk.h deleted file mode 100644 index 3bb309986ed..00000000000 --- a/poros/poros/converter/gpu/topk.h +++ /dev/null @@ -1,59 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file topk.h -* @author tianjinjin@baidu.com -* @date Tue Sep 7 14:29:20 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class TopkConverter : public GpuConverter { -public: - TopkConverter() {} - virtual ~TopkConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::topk(Tensor self, int k, int dim=-1, bool largest=True, bool sorted=True) -> (Tensor values, Tensor indices)", - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::topk.values(Tensor self, int k, int dim=-1, bool largest=True, bool sorted=True, *, Tensor(a!) values, Tensor(b!) indices) -> (Tensor(a!) values, Tensor(b!) indices)", - * **/ - const std::vector node_kind() { - return {torch::jit::aten::topk, - }; - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/unary.cpp b/poros/poros/converter/gpu/unary.cpp deleted file mode 100644 index 71835c0e946..00000000000 --- a/poros/poros/converter/gpu/unary.cpp +++ /dev/null @@ -1,182 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file unary.cpp -* @author tianjinjin@baidu.com -* @date Mon Sep 6 20:23:14 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/unary.h" -#include "poros/converter/gpu/weight.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/engine/tensorrt_engine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/context/poros_global.h" -#include "poros/util/macros.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/* -"aten::cos(Tensor self) -> Tensor", -*/ -bool UnaryConverter::converter(TensorrtEngine* engine, const torch::jit::Node *node) { - at::ArrayRef inputs = node->inputs(); - POROS_CHECK_TRUE((inputs.size() == 1), "invaid inputs size for UnaryConverter"); - if (node->schema().operator_name() != - torch::jit::parseSchema("aten::floor.float(float a) -> (int)").operator_name()) { - POROS_CHECK_TRUE((inputs[0]->type()->isSubtypeOf(c10::TensorType::get())), - "input[0] for UnaryConverter is not Tensor as expected"); - } - - //extract self - auto self = engine->context().get_tensor(inputs[0]); - POROS_CHECK_TRUE((self != nullptr), "Unable to init input tensor for node: " << *node); - - nvinfer1::UnaryOperation trt_type; - switch (node->kind()) { - case torch::jit::aten::cos: - trt_type = nvinfer1::UnaryOperation::kCOS; - break; - case torch::jit::aten::acos: - trt_type = nvinfer1::UnaryOperation::kACOS; - break; - case torch::jit::aten::cosh: - trt_type = nvinfer1::UnaryOperation::kCOSH; - break; - case torch::jit::aten::sin: - trt_type = nvinfer1::UnaryOperation::kSIN; - break; - case torch::jit::aten::asin: - trt_type = nvinfer1::UnaryOperation::kASIN; - break; - case torch::jit::aten::sinh: - trt_type = nvinfer1::UnaryOperation::kSINH; - break; - case torch::jit::aten::tan: - trt_type = nvinfer1::UnaryOperation::kTAN; - break; - case torch::jit::aten::atan: - trt_type = nvinfer1::UnaryOperation::kATAN; - break; - case torch::jit::aten::abs: - trt_type = nvinfer1::UnaryOperation::kABS; - break; - case torch::jit::aten::floor: - trt_type = nvinfer1::UnaryOperation::kFLOOR; - break; - case torch::jit::aten::reciprocal: - trt_type = nvinfer1::UnaryOperation::kRECIP; - break; - case torch::jit::aten::log: - trt_type = nvinfer1::UnaryOperation::kLOG; - break; - case torch::jit::aten::ceil: - trt_type = nvinfer1::UnaryOperation::kCEIL; - break; - case torch::jit::aten::sqrt: - trt_type = nvinfer1::UnaryOperation::kSQRT; - break; - case torch::jit::aten::exp: - trt_type = nvinfer1::UnaryOperation::kEXP; - break; - case torch::jit::aten::neg: - trt_type = nvinfer1::UnaryOperation::kNEG; - break; - case torch::jit::aten::erf: - trt_type = nvinfer1::UnaryOperation::kERF; - break; - case torch::jit::aten::asinh: - trt_type = nvinfer1::UnaryOperation::kASINH; - break; - case torch::jit::aten::acosh: - trt_type = nvinfer1::UnaryOperation::kACOSH; - break; - case torch::jit::aten::atanh: - trt_type = nvinfer1::UnaryOperation::kATANH; - break; - case torch::jit::aten::log2: - trt_type = nvinfer1::UnaryOperation::kLOG; - break; - case torch::jit::aten::log10: - trt_type = nvinfer1::UnaryOperation::kLOG; - break; - case torch::jit::aten::round: - trt_type = nvinfer1::UnaryOperation::kROUND; - break; - default: - POROS_THROW_ERROR("We should never reach here for UnaryConverter, meet Unsupported node kind!"); - } - //IUnaryLayer only support: operation NEG not allowed on type Int32 - nvinfer1::DataType self_type = self->getType(); - const nvinfer1::DataType allowed_type = trt_type == nvinfer1::UnaryOperation::kNOT ? nvinfer1::DataType::kBOOL : nvinfer1::DataType::kFLOAT; - bool should_cast = self_type == allowed_type ? false : true; - if (should_cast) { - nvinfer1::IIdentityLayer* cast_layer = engine->network()->addIdentity(*self); - cast_layer->setName((layer_info(node) + "_IIdentityLayer").c_str()); - cast_layer->setOutputType(0, allowed_type); - self = cast_layer->getOutput(0); - } - - auto unary = engine->network()->addUnary(*self, trt_type); - POROS_CHECK(unary, "Unable to create unary layer from node: " << *node); - unary->setName((layer_info(node) + "_IUnaryLayer").c_str()); - auto output = unary->getOutput(0); - if (trt_type == nvinfer1::UnaryOperation::kLOG) { - nvinfer1::ITensor* alphaTensor = nullptr; - if (node->kind() == torch::jit::aten::log2) { - alphaTensor = tensor_to_const(engine, torch::tensor(std::log2(std::exp(1)), {torch::kFloat32})); - } else if (node->kind() == torch::jit::aten::log10) { - alphaTensor = tensor_to_const(engine, torch::tensor(std::log10(std::exp(1)), {torch::kFloat32})); - } else { - // need not to do anything. - } - // ln(x) * log2(e) = log2(x) - // ln(x) * log10(e) = log10(x) - if (alphaTensor != nullptr) { - auto scaleLayer = add_elementwise(engine, - nvinfer1::ElementWiseOperation::kPROD, - output, - alphaTensor, - layer_info(node) + std::string("_prod")); - POROS_CHECK(scaleLayer, "Unable to create scale layer from node: " << *node); - output = scaleLayer->getOutput(0); - } - } - if (node->schema().operator_name() == - torch::jit::parseSchema("aten::floor.float(float a) -> (int)").operator_name()) { - auto identity = engine->network()->addIdentity(*output); - identity->setOutputType(0, nvinfer1::DataType::kINT32); - identity->setName((layer_info(node) + "_IIdentityLayer_for_output").c_str()); - output = identity->getOutput(0); - } else if (should_cast) { - nvinfer1::IIdentityLayer* castback_layer = engine->network()->addIdentity(*output); - castback_layer->setName((layer_info(node) + "_IIdentityLayer_for_output").c_str()); - castback_layer->setOutputType(0, self_type); - output = castback_layer->getOutput(0); - } - engine->context().set_tensor(node->outputs()[0], output); - LOG(INFO) << "Output tensor shape: " << output->getDimensions(); - return true; -} - -POROS_REGISTER_CONVERTER(TensorrtEngine, UnaryConverter); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/unary.h b/poros/poros/converter/gpu/unary.h deleted file mode 100644 index b5604e68693..00000000000 --- a/poros/poros/converter/gpu/unary.h +++ /dev/null @@ -1,112 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file unary.h -* @author tianjinjin@baidu.com -* @date Mon Sep 6 20:23:14 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" - -#include "poros/converter/gpu/gpu_converter.h" -#include "poros/engine/tensorrt_engine.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class UnaryConverter : public GpuConverter { -public: - UnaryConverter() {} - virtual ~UnaryConverter() {} - - bool converter(TensorrtEngine* engine, const torch::jit::Node *node); - - const std::vector schema_string() { - return {"aten::cos(Tensor self) -> Tensor", - "aten::acos(Tensor self) -> Tensor", - "aten::cosh(Tensor self) -> Tensor", - "aten::sin(Tensor self) -> Tensor", - "aten::asin(Tensor self) -> Tensor", - "aten::sinh(Tensor self) -> Tensor", - "aten::tan(Tensor self) -> Tensor", - "aten::atan(Tensor self) -> Tensor", - "aten::abs(Tensor self) -> Tensor", - "aten::floor(Tensor self) -> Tensor", - "aten::reciprocal(Tensor self) -> Tensor", - "aten::log(Tensor self) -> Tensor", - "aten::ceil(Tensor self) -> Tensor", - "aten::sqrt(Tensor self) -> Tensor", - "aten::exp(Tensor self) -> Tensor", - "aten::neg(Tensor self) -> Tensor", - "aten::erf(Tensor self) -> Tensor", - "aten::asinh(Tensor self) -> Tensor", - "aten::acosh(Tensor self) -> Tensor", - "aten::atanh(Tensor self) -> Tensor", - "aten::log2(Tensor self) -> (Tensor)", - "aten::log10(Tensor self) -> (Tensor)", - "aten::floor.float(float a) -> (int)", - "aten::round(Tensor self) -> (Tensor)" - }; - } - - /** TODO: TO SUPPORT CONVERTERS BELLOW: - * "aten::cos.out(Tensor self, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::acos.out(Tensor self, *, Tensor(a!) out) -> Tensor(a!)", - * "aten::cosh.out(Tensor self, *, Tensor(a!) out) -> Tensor(a!)", - * "ALL OF THEIR .out CONVERTERS IS NOT SUPPORTED" - * "ALL OF THEIR .out CONVERTERS IS NOT SUPPORTED" - * "ALL OF THEIR .out CONVERTERS IS NOT SUPPORTED" - * **/ - const std::vector node_kind() { - return {torch::jit::aten::cos, - torch::jit::aten::acos, - torch::jit::aten::cosh, - torch::jit::aten::sin, - torch::jit::aten::asin, - torch::jit::aten::sinh, - torch::jit::aten::tan, - torch::jit::aten::atan, - torch::jit::aten::abs, - torch::jit::aten::floor, - torch::jit::aten::reciprocal, - torch::jit::aten::log, - torch::jit::aten::ceil, - torch::jit::aten::sqrt, - torch::jit::aten::exp, - torch::jit::aten::neg, - torch::jit::aten::erf, - torch::jit::aten::asinh, - torch::jit::aten::acosh, - torch::jit::aten::atanh, - torch::jit::aten::log2, - torch::jit::aten::log10, - torch::jit::aten::round - }; - } - bool assign_schema_attr() { - return assign_schema_attr_helper({{"aten::floor.float(float a) -> (int)", {1, 1}}}); - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/weight.cpp b/poros/poros/converter/gpu/weight.cpp deleted file mode 100644 index 1244acccb75..00000000000 --- a/poros/poros/converter/gpu/weight.cpp +++ /dev/null @@ -1,111 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file weight.cpp -* @author tianjinjin@baidu.com -* @date Fri Aug 6 14:17:11 CST 2021 -* @brief -**/ - -#include "poros/converter/gpu/weight.h" -#include "poros/engine/trtengine_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -Weights::Weights() { - this->inputs_num = 0; - this->outputs_num = 0; - this->data.type = nvinfer1::DataType::kFLOAT; - this->data.values = nullptr; - this->data.count = 0; -} - -Weights::Weights(at::Tensor tensor) { - POROS_CHECK((tensor.sizes().size() <= nvinfer1::Dims::MAX_DIMS), - "given tensor is outof max_dims"); - - if (tensor.scalar_type() == c10::ScalarType::Long) { - LOG(WARNING) << "Weights meets c10::ScalarType::Long tensor type, change this to c10::ScalarType::Int. " - << "Attention: this may leed to percision change"; - tensor = tensor.to(at::ScalarType::Int); - } - - this->shape = sizes_to_nvdim(tensor.sizes()); - //TODO: CHECK this bias info. - this->inputs_num = (tensor.sizes().size() > 1) ? tensor.sizes()[1] : tensor.sizes()[0]; - this->outputs_num = tensor.sizes()[0]; - - if (tensor.sizes().size() > 2) { - this->kernel_shape.nbDims = tensor.sizes().size() - 2; - for (size_t i = 2; i < tensor.sizes().size(); i++) { - this->kernel_shape.d[i - 2] = tensor.sizes()[i]; - } - } else { - this->kernel_shape.nbDims = 1; - this->kernel_shape.d[0] = 1; - } - - auto t_cpu = tensor.to(at::kCPU); - t_cpu = t_cpu.contiguous(); - - auto t_type = c10::optTypeMetaToScalarType(t_cpu.dtype()); - POROS_CHECK(t_type.has_value(), "unsupported datatype"); - //TODO: may be failed here - auto dtype = attype_to_nvtype(t_type.value()); - - void* buf = nullptr; - if (dtype == nvinfer1::DataType::kFLOAT) { - buf = malloc(t_cpu.numel() * sizeof(float)); - memcpy(buf, t_cpu.data_ptr(), t_cpu.numel() * sizeof(float)); - } else if (dtype == nvinfer1::DataType::kHALF) { - buf = malloc(t_cpu.numel() * (sizeof(float) / 2)); - memcpy(buf, t_cpu.data_ptr(), t_cpu.numel() * (sizeof(float) / 2)); - } else if (dtype == nvinfer1::DataType::kINT8) { - buf = malloc(t_cpu.numel() * sizeof(char)); - memcpy(buf, t_cpu.data_ptr(), t_cpu.numel() * sizeof(char)); - } else if (dtype == nvinfer1::DataType::kINT32) { - buf = malloc(t_cpu.numel() * sizeof(int)); - memcpy(buf, t_cpu.data_ptr(), t_cpu.numel() * sizeof(int)); - } else if (dtype == nvinfer1::DataType::kBOOL) { - buf = malloc(t_cpu.numel() * sizeof(bool)); - memcpy(buf, t_cpu.data_ptr(), t_cpu.numel() * sizeof(bool)); - } - - this->data.type = dtype; - this->data.count = t_cpu.numel(); - this->data.values = buf; - -} - -std::ostream& operator<<(std::ostream& os, const Weights& w) { - os << "Weights: " << w.shape - << "\n Number of input maps: " << w.inputs_num - << "\n Number of output maps: " << w.outputs_num - << "\n Element shape: ["; - for (int i = 0; i < w.kernel_shape.nbDims; i++) { - os << w.kernel_shape.d[i]; - if (i + 1 < w.kernel_shape.nbDims) { - os << ','; - } - } - os << ']'; - return os; -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/gpu/weight.h b/poros/poros/converter/gpu/weight.h deleted file mode 100644 index 13b6e90c811..00000000000 --- a/poros/poros/converter/gpu/weight.h +++ /dev/null @@ -1,68 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file weight.h -* @author tianjinjin@baidu.com -* @date Fri Aug 13 11:14:51 CST 2021 -* @brief -**/ - -#pragma once - -#include - -#include "torch/script.h" -#include "NvInfer.h" - -#include "poros/engine/tensorrt_engine.h" -#include "poros/util/macros.h" - -namespace baidu { -namespace mirana { -namespace poros { - -struct Weights { - nvinfer1::Weights data; - nvinfer1::Dims kernel_shape; - nvinfer1::Dims shape; - int64_t inputs_num; - int64_t outputs_num; - - Weights(); - Weights(at::Tensor tensor); - // Weights(float val); - // Weights(int32_t val); - friend std::ostream& operator<<(std::ostream& os, const Weights& w); -}; - -inline nvinfer1::ITensor* tensor_to_const(TensorrtEngine* engine, at::Tensor t) { - auto t_weights = Weights(t); - auto const_layer = engine->network()->addConstant(t_weights.shape, t_weights.data); - POROS_CHECK(const_layer, "unable to freeze tensor to constant"); - - auto out = const_layer->getOutput(0); - - std::ostringstream tensor_id; - tensor_id << reinterpret_cast(out); - - LOG(INFO) << "Freezing tensor " << tensor_id.str() << " as an IConstantLayer"; - const_layer->setName(("[Freeze Tensor " + tensor_id.str() + " ]").c_str()); - - return out; -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/converter/iconverter.h b/poros/poros/converter/iconverter.h deleted file mode 100644 index 58ec34b1733..00000000000 --- a/poros/poros/converter/iconverter.h +++ /dev/null @@ -1,424 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file iconverter.h -* @author tianjinjin@baidu.com -* @author huangben@baidu.com -* @date Tue Jul 27 11:24:21 CST 2021 -* @brief -**/ - -#pragma once - -#include - -#include "torch/script.h" -#include "ATen/core/function_schema.h" -#include "torch/csrc/jit/frontend/function_schema_parser.h" -#include "torch/csrc/jit/ir/ir.h" -#include "torch/csrc/jit/runtime/custom_operator.h" - -#include "poros/context/poros_global.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class IEngine; - -// schema属性 -// 默认支持dynamic shape,不支持tensor scalar输入 -struct schema_attr { - int is_support_dynamic_shape = 1; - int is_support_tensor_scalar = 0; -}; - -class IConverter { -public: - virtual ~IConverter() {} - /** - * @brief converter核心实现,注意要将输入tensor和engine的tensor对应起来,通过engine.context - * tensor->tensor, constant->constant - * @param [in] sub_graph : 子图 - * @return [res]int - * @retval 0 => success, <0 => fail - **/ - virtual bool converter(IEngine* engine, const torch::jit::Node *node) = 0; - virtual const std::vector schema_string() = 0; - virtual const std::vector node_kind() = 0; - const std::unordered_map get_schema_attr_map() { - return _schema_attr_map; - } -protected: - /** - * @brief help assign schema attr - * - * @param [in] schema_attr_vec : schema and schema_attr which want to assign - * @return true => succeed false => failed - **/ - virtual bool assign_schema_attr_helper(std::vector> schema_attr_vec) { - if (_schema_attr_map.empty()) { - LOG(INFO) << "the schema_attr_map may not have been initialized."; - return false; - } - for (size_t i = 0; i < schema_attr_vec.size(); i++) { - c10::OperatorName op_name = torch::jit::parseSchema(schema_attr_vec[i].first).operator_name(); - if (_schema_attr_map.count(op_name) == 0) { - LOG(INFO) << "schema: [ " << schema_attr_vec[i].first << " ] was not found in schema_attr_map"; - return false; - } - _schema_attr_map[op_name] = schema_attr_vec[i].second; - } - return true; - } - // 给schema赋予attr,在子类中实现 - virtual bool assign_schema_attr() { - return true; - } -private: - // 声明 ConvertersMap为友元,在其中调用init_schema_attr - friend class ConvertersMap; - // 初始化schema attr,需converter为子类时调用 - bool init_schema_attr() { - _schema_attr_map.clear(); - std::vector schema_strings = this->schema_string(); - schema_attr attr; - for (const std::string& s : schema_strings) { - _schema_attr_map.insert({torch::jit::parseSchema(s).operator_name(), attr}); - } - return assign_schema_attr(); - } - std::unordered_map _schema_attr_map; -}; - -struct ConverterOptions { - std::vector valid_schemas; - - ConverterOptions() = default; - - ConverterOptions& set_valid_schemas(std::vector schema_string) { - use_options = true; - for (auto s : schema_string) { - valid_schemas.push_back(torch::jit::parseSchema(s).operator_name()); - } - return *this; - } - - bool use() { - return use_options; - } -private: - bool use_options = false; -}; - -struct ConvRegistration { - torch::jit::NodeKind kind; - IConverter* converter; - ConverterOptions options; -}; - -class ConvertersMap { -public: - ConvertersMap() {} - - virtual ~ConvertersMap() { - } - - //添加converter到当前的map。 - bool add_converter(torch::jit::NodeKind node_kind, ConvRegistration conv_reg) { - auto iter = converters_map.find(node_kind); - if (iter != converters_map.end()) { - LOG(WARNING) << "override converter for [ " << node_kind.toQualString() << " ]"; - } - converters_map[node_kind] = std::move(conv_reg); - return true; - } - - IConverter* get_converter(const torch::jit::Node* node) { - if (!node_converterable(node)) { - return nullptr; - } - auto node_kind = node->kind(); - auto iter = converters_map.find(node_kind); - if (iter == converters_map.end()) { - return nullptr; - } - auto conv_reg = iter->second; - if (conv_reg.options.use()) { - if (conv_reg.options.valid_schemas.size() != 0) { - auto schema = node->maybeSchema(); - if (!schema) { - return nullptr; - } - for (auto reg_schema : conv_reg.options.valid_schemas) { - if (reg_schema == schema->operator_name()) { - return conv_reg.converter; - } - } - return nullptr; - } - } - return conv_reg.converter;; - } - - // 判断list类型的输入输出长度是否发生变化 - bool list_size_is_variable_length(const torch::jit::Node *node) { - auto list_size_map_input = PorosGlobalContext::instance()._list_size_map._list_size_map_input; - for (size_t i = 0; i < node->inputs().size(); i++) { - auto value = node->input(i); - // 如果是list类型 - if (value->type()->kind() == c10::TypeKind::ListType) { - if (list_size_map_input.count(value) != 0 && list_size_map_input[value].count(const_cast(node)) != 0) { - // 如果本node对应value(即list变量)记录的size有1个以上的,说明长度发生变化 - // 返回true,外部no-converterable - if (list_size_map_input[value].at(const_cast(node)).size() != 1){ - return true; - } - } - } - } - // 输出list变量判断长度是否变化,原理同输入 - auto list_size_map_output = PorosGlobalContext::instance()._list_size_map._list_size_map_output; - for (size_t i = 0; i < node->outputs().size(); i++) { - auto value = node->output(i); - if (value->type()->kind() == c10::TypeKind::ListType) { - if (list_size_map_output.count(value) != 0 && list_size_map_output[value].count(const_cast(node)) != 0) { - if (list_size_map_output[value].at(const_cast(node)).size() != 1){ - return true; - } - } - } - } - return false; - } - // 判断特殊输出类型节点例如list[list[]] - bool special_node_check(const torch::jit::Node *node) { - if (node->kind() == torch::jit::prim::ListConstruct) { - const torch::jit::Value* output = node->outputs()[0]; - if (output->type()->str().find("[][]") != output->type()->str().npos) { - return true; - } - } - return false; - } - - // 判断是否属于支持tensor scalar输入的op集合 - bool is_unsupport_tensor_scalar_inputs(const torch::jit::Node *node, - std::unordered_map schema_attr_map) { - if (node->kind() == torch::jit::prim::CudaFusionGroup) { - return false; - } - for(size_t i = 0; i < node->inputs().size(); i++) { - // 如果input是scalar或scalar list,且不来自于prim::Constant - torch::jit::Value* current_input = node->input(i); - if ((current_input->type()->isSubtypeOf(c10::NumberType::get()) || - current_input->type()->isSubtypeOf(c10::BoolType::get()) || - current_input->type()->isSubtypeOf(c10::StringType::get()) || - current_input->type()->isSubtypeOf(c10::ListType::ofFloats()) || - current_input->type()->isSubtypeOf(c10::ListType::ofInts()) || - current_input->type()->isSubtypeOf(c10::ListType::ofBools()) || - current_input->type()->isSubtypeOf(c10::ListType::ofStrings()) - ) && - current_input->node()->kind() != torch::jit::prim::Constant) { - // 判断node是否属于: 1、prim::ListConstruct 2、prim::ListUnpack 3、支持scalar tensor输入的schema - // 都不属于则不支持,返回true,外部no-converterable - if (!node->maybeSchema()) { - // 这两个op没有schema,需要单独判断 - if (node->kind() == torch::jit::prim::ListConstruct || - node->kind() == torch::jit::prim::ListUnpack) { - return false; - } else { - return true; - } - } else { - if (schema_attr_map[node->maybeSchema()->operator_name()].is_support_tensor_scalar == 1) { - return false; - } else { - return true; - } - } - } - } - return false; - } - - bool node_converterable(const torch::jit::Node* node) { - auto node_kind = node->kind(); - auto iter = converters_map.find(node_kind); - if (iter == converters_map.end()) { - LOG(WARNING) << "no converter find for [ " << node_kind.toQualString() << " ]"; - if (node->maybeSchema()) { - LOG(WARNING) << "unsupported schema is [ " << *node->maybeSchema() << " ]"; - } - return false; - } - auto conv_reg = iter->second; - if (conv_reg.options.use()) { - if (conv_reg.options.valid_schemas.size() != 0) { - auto schema = node->maybeSchema(); - if (!schema) { - LOG(WARNING) << "no schema find for [ " << node_kind.toQualString() << " ]"; - return false; - } - // 检查用户自定义不支持node schema - if (_unsupport_schema_set.count(schema->operator_name())) { - LOG(WARNING) << "The user specifies that the unsupported node schema is [ " << *schema << " ]"; - return false; - } - // 检查用户自定义不支持node kind - if (_unsupport_nodekind_set.count(node->kind())) { - LOG(WARNING) << "The user specifies that the unsupported node kind is [ " << node->kind().toQualString() << " ]"; - return false; - } - // 由于tensorrt支持问题,aten::_convolution为反卷积时(transposed==true) - // output_padding参数必须为0,否则不支持 - if (node->kind() == torch::jit::aten::_convolution && node->inputs().size() >= 12) { - if (node->input(6)->node()->kind() == torch::jit::prim::Constant && - node->input(6)->type()->kind() == c10::TypeKind::BoolType && - node->input(7)->node()->kind() == torch::jit::prim::Constant && - node->input(7)->type()->isSubtypeOf(c10::ListType::ofInts())) { - - bool transposed = toIValue(node->input(6)).value().toBool(); - auto input_7_vec = toIValue(node->input(7)).value().toIntVector(); - if (transposed && (input_7_vec[0] > 0 || input_7_vec[1] > 0)) { - LOG(INFO) << "TensorRT does not have a notion of output_padding for deconvolution layers." - " output_padding has to be set as zeros."; - return false; - } - } - } - IConverter* current_converter = conv_reg.converter; - if (!current_converter->init_schema_attr()) { - LOG(WARNING) << "converter [ " << node_kind.toQualString() << " ] failed to initialize schema attribute."; - return false; - } - auto conv_schema_attr = current_converter->get_schema_attr_map(); - if (conv_schema_attr.count(schema->operator_name()) == 0) { - LOG(WARNING) << "no supported schema find for [ " << node_kind.toQualString() << " ]"; - LOG(WARNING) << "unsupported schema is [ " << *schema << " ]"; - return false; - } - - // 如果是dynamic,检查no_supported_dynamic_schema - PorosOptions poros_options = PorosGlobalContext::instance().get_poros_options(); - if (poros_options.is_dynamic) { - if (conv_schema_attr[schema->operator_name()].is_support_dynamic_shape == 0) { - LOG(WARNING) << "no supported dynamic schema is [ " << *schema << " ]"; - return false; - } - } - - auto node_kind = node->kind(); - // 特殊op的特殊判断,比如list[list[]]在ListConstructConverter中的支持问题 - if (special_node_check(node)) { - LOG(WARNING) << "node input or output type is not support: " << node_kind.toQualString(); - return false; - } - - // list输入输出长度是否变化 - if (list_size_is_variable_length(node)) { - LOG(WARNING) << "input or output is variable length list [ " << node_kind.toQualString() << " ]"; - return false; - } - - // scalar的问题判断,判断一个op是否支持把scalar变成tensor来使用 - if (is_unsupport_tensor_scalar_inputs(node, conv_schema_attr)) { - LOG(WARNING) << "unsupport nvtensor scalar input node is [ " << node_kind.toQualString() << " ]"; - return false; - } - - // 如果是mutable的op,是否属于目前支持的范围 - if (node->kind().is_aten() && node->schema().is_mutable() && node->input(0)->type()->kind() == c10::TypeKind::ListType) { - if (PorosGlobalContext::instance().supported_mutable_ops_set.count(node->kind()) == 0) { - LOG(WARNING) << "Meet unsupport mutable node. The node is [ " << node_kind.toQualString() << " ]"; - return false; - } - } - } else { - // 用户指定不支持OP列表时,图中node只有kind没有schema的,只比较nodekind - if (_unsupport_nodekind_set.count(node->kind())) { - LOG(WARNING) << "The user specifies that the unsupported node kind is [ " << node->kind().toQualString() << " ]"; - return false; - } - } - } - return true; - } - // 由全局option初始化不支持op set - void init_unsupport_op_set() { - try { - std::vector unsupport_op_vec = PorosGlobalContext::instance().get_poros_options().unsupport_op_list; - // 每次设置unsupport_op_list时刷新schema和nodekind set,避免用户下次编译想重新设置还保留之前不支持的op - _unsupport_schema_set.clear(); - _unsupport_nodekind_set.clear(); - for (size_t i = 0; i < unsupport_op_vec.size(); i++) { - std::string line = unsupport_op_vec[i]; - if (line.size() == 0) { - continue; - } - auto schema_or_opname = torch::jit::parseSchemaOrName(line); - // operator name - if (schema_or_opname.is_left()) { - torch::jit::NodeKind node_kind = c10::Symbol::fromQualString(line); - if (!converters_map.count(node_kind)) { - LOG(WARNING) << "WARNING: The user-defined unsupported nodekind [ " << node_kind.toQualString() << " ] cannot be found in the poros supported op set." - " Please check the PorosOptions.unsupport_op_list input."; - } - _unsupport_nodekind_set.insert(node_kind); - LOG(INFO) << "The user-defined unsupported node kind is [ " << node_kind.toQualString() << " ]."; - // schema - } else { - c10::FunctionSchema fs = schema_or_opname.right(); - std::string fs_name = fs.name(); - auto end_index = fs_name.find_last_of('.') == std::string::npos ? fs_name.size() : fs_name.find_last_of('.'); - std::string node_kind_name = fs_name.substr(0, end_index); - torch::jit::NodeKind node_kind = c10::Symbol::fromQualString(node_kind_name); - c10::OperatorName node_op_name = schema_or_opname.right().operator_name(); - if (converters_map.count(node_kind)) { - std::vector node_valid_schema_vec = converters_map[node_kind].options.valid_schemas; - int e = 0; - for (auto i : node_valid_schema_vec) { - if (i == node_op_name) { - e++; - } - } - if (!e) { - LOG(WARNING) << "WARNING: The user-defined unsupported schema [ " << line << " ] cannot be found in the poros supported op set." - " Please check the PorosOptions.unsupport_op_list input."; - } - } else { - LOG(WARNING) << "WARNING: The user-defined unsupported schema nodekind [ " << node_kind.toQualString() << " ] cannot be found in the poros supported op set." - " Please check the PorosOptions.unsupport_op_list input."; - } - _unsupport_schema_set.insert(node_op_name); - LOG(INFO) << "The user-defined unsupported schema is [ " << fs << " ]."; - } - } - } catch (...) { - LOG(WARNING) << "WARNING: Failed to initialize user-defined unsupport operator list. Please check the PorosOptions.unsupport_op_list parameter."; - } - } - -private: - std::set converter_schemas; - std::unordered_map converters_map; - std::unordered_set _unsupport_nodekind_set; - std::unordered_set _unsupport_schema_set; - -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/engine/engine.cpp b/poros/poros/engine/engine.cpp deleted file mode 100644 index 9b1c55715db..00000000000 --- a/poros/poros/engine/engine.cpp +++ /dev/null @@ -1,52 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file engine.cpp -* @author huangben@baidu.com -* @date Mon Mar 8 11:36:11 CST 2021 -* @brief -**/ - -#include "poros/engine/iengine.h" -#include "poros/context/poros_global.h" -#include "poros/converter/iconverter.h" - -namespace baidu { -namespace mirana { -namespace poros { - -bool IEngine::is_node_supported(const torch::jit::Node* node) { - auto converter_map = PorosGlobalContext::instance().get_converter_map(who_am_i()); - if (converter_map != nullptr && converter_map->node_converterable(node)) { - return true; - } else { - if (node->kind() != torch::jit::prim::Loop && - node->kind() != torch::jit::prim::If && - node->kind() != torch::jit::prim::CudaFusionGroup && - node->kind() != torch::jit::prim::Param) { - LOG(INFO) << "not supported node: " << node->kind().toQualString() - << ", detail info: " << *node; - } - // auto convertableItr = get_non_convertable_nodes().find(node->kind().toQualString()); - // if (convertableItr != get_non_convertable_nodes().end()) { - // return true; - // } - return false; - } -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/engine/engine_context.h b/poros/poros/engine/engine_context.h deleted file mode 100644 index cf60dffc059..00000000000 --- a/poros/poros/engine/engine_context.h +++ /dev/null @@ -1,100 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file engine_context.h -* @author tianjinjin@baidu.com -* @date Fri Jul 23 11:21:10 CST 2021 -* @brief -**/ - -#pragma once - -#include -#include -#include "torch/script.h" - -namespace baidu { -namespace mirana { -namespace poros { - -template -class EngineContext { -public: - explicit EngineContext() {} - - T* get_tensor(const torch::jit::Value* value) { - auto it = _value_tensor_map.find(value); - if (it == _value_tensor_map.end()) { - return nullptr; - } - return it->second; - } - - bool set_tensor(const torch::jit::Value* value, T* tensor) { - if (value != nullptr && tensor != nullptr) { - _value_tensor_map[value] = tensor; - return true; - } - return false; - } - - bool get_tensorlist(const torch::jit::Value* value, std::vector& tensorlist) { - auto it = _value_tensorlist_map.find(value); - if (it == _value_tensorlist_map.end()) { - return false; - } - tensorlist = it->second; - return true; - } - - bool set_tensorlist(const torch::jit::Value* value, std::vector tensorlist) { - if (value != nullptr) { - _value_tensorlist_map[value] = tensorlist; - return true; - } - return false; - } - - torch::jit::IValue get_constant(const torch::jit::Value* value) { - auto it = _value_constant_map.find(value); - if (it != _value_constant_map.end()) { - return it->second; - } else { - return torch::jit::IValue(); - } - } - - bool set_constant(const torch::jit::Value* value, torch::jit::IValue constant) { - if (value != nullptr) { - _value_constant_map[value] = constant; - return true; - } - return false; - } - -private: - std::string _engine_id; - //value <-> nvtensor - std::unordered_map _value_tensor_map; - //value <-> nvtensor list - //std::unordered_map> _value_tensorlist_map; - std::unordered_map> _value_tensorlist_map; - //value <-> others - std::unordered_map _value_constant_map; -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/engine/iengine.h b/poros/poros/engine/iengine.h deleted file mode 100644 index a9d1094bc24..00000000000 --- a/poros/poros/engine/iengine.h +++ /dev/null @@ -1,91 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file iengine.h -* @author tianjinjin@baidu.com -* @author huangben@baidu.com -* @date Mon Mar 8 11:36:11 CST 2021 * @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "torch/script.h" -#include "torch/csrc/jit/ir/ir.h" -#include "ATen/core/interned_strings.h" - -#include "poros/iplugin/plugin_create.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/** - * the base engine class - * every registered engine should inherit from this IEngine - **/ - -struct PorosGraph { - torch::jit::Graph* graph = NULL; - torch::jit::Node* node = NULL; -}; - -typedef uint64_t EngineID; - -class IEngine : public IPlugin, public torch::CustomClassHolder{ -public: - virtual ~IEngine() {} - - /** - * @brief init, 必须init成功才算初始化成功 - * @return int - * @retval 0 => success, <0 => fail - **/ - virtual int init() = 0; - - /** - * @brief 编译期将subgraph转化成对应engine的图结构保存在engine内部,以使得运行期的excute_engine能调用, 此处保证所有的op都被支持,核心实现 - * @param [in] sub_graph : 子图 - * @return [res]int - * @retval 0 => success, <0 => fail - **/ - virtual int transform(const PorosGraph& sub_graph) = 0; - - /** - * @brief 子图执行期逻辑 - * @param [in] inputs : 输入tensor - * @return [res] 输出tensor - **/ - virtual std::vector excute_engine(const std::vector& inputs) = 0; - - virtual void register_module_attribute(const std::string& name, torch::jit::Module& module) = 0; - - //标识 - virtual const std::string who_am_i() = 0; - - //node是否被当前engine支持 - bool is_node_supported(const torch::jit::Node* node); - -public: - std::pair _num_io; //输入/输出参数个数 - EngineID _id; - -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/engine/tensorrt_engine.cpp b/poros/poros/engine/tensorrt_engine.cpp deleted file mode 100644 index 650db10033f..00000000000 --- a/poros/poros/engine/tensorrt_engine.cpp +++ /dev/null @@ -1,515 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file tensorrt_engine.cpp -* @author tianjinjin@baidu.com -* @author huangben@baidu.com -* @date Mon Mar 8 11:36:11 CST 2021 -* @brief -**/ -#include "poros/engine/tensorrt_engine.h" - -#include "poros/context/poros_global.h" -#include "poros/converter/gpu/converter_util.h" -#include "poros/converter/iconverter.h" -// #include "poros/engine/trtengine_util.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - - -TensorrtEngine::TensorrtEngine() : _logger(get_nvlogger().torch_level()), _builder(nullptr), - _network(nullptr), _cfg(nullptr), _runtime(nullptr), _cuda_engine(nullptr), - _exec_ctx(nullptr), _mutex(nullptr) { - // init nvinfer plgins - initLibNvInferPlugins(&_logger, ""); -} - -TensorrtEngine::TensorrtEngine(std::string engine_str) : _logger(get_nvlogger().torch_level()) { - - init(); - - _cuda_engine = make_shared_ptr(_runtime->deserializeCudaEngine((void*)engine_str.c_str(), engine_str.length())); - _exec_ctx = make_shared_ptr(_cuda_engine->createExecutionContext()); - binding_io(); -} - -//TensorrtEngine::~TensorrtEngine() { -// //_exec_ctx->destroy(); -// //_cuda_engine->destroy(); -// //_runtime->destroy(); -// -//} - -void TensorrtEngine::binding_io() { - uint64_t inputs = 0; - uint64_t outputs = 0; - - for (int64_t idx = 0; idx < _cuda_engine->getNbBindings(); idx++) { - std::string name = _cuda_engine->getBindingName(idx); - //if (name.find("profile") != name.npos) { - // continue; - //} - std::string idx_s = name.substr(name.find("_") + 1); - uint64_t idx_new = static_cast(std::stoi(idx_s)); - if (_cuda_engine->bindingIsInput(idx)) { - inputs++; - _in_binding_map[idx] = idx_new; - } else { - outputs++; - _out_binding_map[idx] = idx_new; - } - } - _num_io = std::make_pair(inputs, outputs); -} - -int TensorrtEngine::init() { - // init nvinfer plgins - initLibNvInferPlugins(&_logger, ""); - _mutex = std::make_shared(); - _poros_options = PorosGlobalContext::instance().get_poros_options(); - - _builder = make_shared_ptr(nvinfer1::createInferBuilder(_logger)); - _network = make_shared_ptr(_builder->createNetworkV2(1U << static_cast( - nvinfer1::NetworkDefinitionCreationFlag::kEXPLICIT_BATCH))); - _cfg = make_shared_ptr(_builder->createBuilderConfig()); - _runtime = make_shared_ptr(nvinfer1::createInferRuntime(_logger)); - - - // Nvidia tf32 is enabled by default. - // if don't want to ues, the BuilderFlag::kTF32 should be clear. - if (!_poros_options.use_nvidia_tf32) { - _cfg->clearFlag(nvinfer1::BuilderFlag::kTF32); - } - if (_poros_options.use_fp16) { - _cfg->setFlag(nvinfer1::BuilderFlag::kFP16); - } -#if NV_TENSORRT_MAJOR >=8 && NV_TENSORRT_MINOR >=3 - // trt version >= 8.3 - _cfg->setMemoryPoolLimit(nvinfer1::MemoryPoolType::kWORKSPACE, _poros_options.max_workspace_size); -#else - _cfg->setMaxWorkspaceSize(_poros_options.max_workspace_size); -#endif - return 0; -} - - -int TensorrtEngine::transform(const PorosGraph& sub_graph) { - //step1. get the given graph - torch::jit::Graph* to_trans_graph = sub_graph.graph; - //PorosGraph to_trans_sub_graph = {to_trans_graph.get(), sub_graph.node}; - - //step2. init the engine input - if (init_engine_inputs(sub_graph) < 0) { - LOG(ERROR) << " init engine inputs failed"; - return -1; - } - - //step3. get op converter_map that tensortengine supports - ConvertersMap* converter_map = PorosGlobalContext::instance().get_converter_map(who_am_i()); - if (converter_map == nullptr) { - LOG(ERROR) << "could not find given engine [ " << who_am_i() << " ] in global context"; - return -1; - } - - //step4. converter the nodes in the given graph one by one. this is the core function. - const torch::jit::Block* block = to_trans_graph->block(); - for (const torch::jit::Node* node : block->nodes()) { - IConverter* conv = converter_map->get_converter(node); - if (nullptr == conv) { - LOG(ERROR) << "pre judgment failed: " << node_info(node); - return -1; - } - - LOG(INFO) << "start to converter for: " << node_info(node); - if (!conv->converter(this, node)) { - LOG(ERROR) << "converter for node failed [ " << *node->maybeSchema() << " ]"; - return -1; - } - } - - //step5. mark the graph output. - at::ArrayRef graph_outputs = block->outputs(); - if (mark_graph_outputs(graph_outputs) < 0) { - LOG(ERROR) << " mark graph outputs failed"; - return -1; - } - - //step6. build cuda engine. - _cuda_engine = make_shared_ptr(_builder->buildEngineWithConfig(*_network, *_cfg)); - if (!_cuda_engine) { - LOG(ERROR) << "build tensorrt engine failed"; - return -1; - } - - //step7. create execution context and binding io - // Easy way to get a unique name for each engine, maybe there is a more - // descriptive way (using something associated with the graph maybe) - _id = reinterpret_cast(_cuda_engine.get()); - _exec_ctx = make_shared_ptr(_cuda_engine->createExecutionContext()); - binding_io(); - - return 0; -} - -//DEPRECATED -inline void TensorrtEngine::gen_tensorrt_input_type(const torch::jit::Value* input, - nvinfer1::DataType& input_type) { - if (input->type()->isSubtypeOf(c10::BoolType::get())) { - input_type = nvinfer1::DataType::kBOOL; - //NumberTypes below - } else if (input->type()->isSubtypeOf(c10::IntType::get())) { - input_type = nvinfer1::DataType::kINT32; - } else if (input->type()->isSubtypeOf(c10::FloatType::get())) { - input_type = nvinfer1::DataType::kFLOAT; - } else { - //TODO: TO ADD LOGGER - } -} - -nvinfer1::Dims TensorrtEngine::gen_dynamic_dims(torch::jit::Value* value) { - if (PorosGlobalContext::instance()._value_dynamic_shape_map.count(value) <= 0) { - LOG(ERROR) << "value is not in value_dynamic_shape_map"; - throw std::runtime_error("value is not in value_dynamic_shape_map"); - } - std::vector sizes = PorosGlobalContext::instance()._value_dynamic_shape_map[value].sizes; - return sizes_to_nvdim(sizes); -} - -//try to extract input value from subgraph_node -int TensorrtEngine::init_engine_inputs(const PorosGraph& sub_graph) { - torch::jit::Node* subgraph_node = sub_graph.node; - torch::jit::Graph* subgraph = sub_graph.graph; - AT_ASSERT(subgraph_node->kind() == torch::jit::prim::CudaFusionGroup); - at::ArrayRef graph_inputs = subgraph->inputs(); - at::ArrayRef node_inputs = subgraph_node->inputs(); - - nvinfer1::IOptimizationProfile* profile = _builder->createOptimizationProfile(); - - bool total_is_dynamic = false; - for (size_t i = 0; i < graph_inputs.size(); i++) { - torch::jit::Value* in = graph_inputs[i]; - torch::jit::Value* node_in = node_inputs[i]; - std::string name = std::string("input_") + std::to_string(i); - - nvinfer1::DataType nv_type; - if (!gen_tensor_type(*subgraph_node, i, nv_type)) { - LOG(WARNING) << "init_engine_inputs failed: reason: can't gen nv_type info from input"; - return -1; - } - - //根据当前poros的设计,subgraph的输入在子图分割阶段,已经全部转换为tensor - //此处如果出现了非tensor类型,则不在poros预期内,不做处理。 - if (in->type()->isSubtypeOf(c10::TensorType::get()) == false) { - LOG(WARNING) << "not supported input type by tensorrt: " << node_info(in->node()); - return -1; - } - - std::vector sizes; - if (!gen_dims_for_tensor(in, sizes)) { - LOG(WARNING) << "gen_dims_for_tensor failed for: " << in->debugName(); - return -1; - }; - nvinfer1::Dims nv_dims = sizes_to_nvdim(sizes); - bool current_dynamic = false; - if (std::find(sizes.begin(), sizes.end(), -1) != sizes.end() || input_is_dynamic(node_in)) { - total_is_dynamic = true; - current_dynamic = true; - } - // mark: 从nv提供的api无法先验地去判断是否是shape tensor - // 这里先这样判断输入的tensor scalar是否属于shape tensor范围 - // 可能会有误判 - bool is_shape_tensor = false; - if (nv_type == nvinfer1::DataType::kINT32 && nv_dims.nbDims <= 1 && - node_in->node()->kind() == torch::jit::aten::tensor) { - torch::jit::use_list in_use_list = in->uses(); - for (size_t u = 0; u < in_use_list.size(); u++) { - if (in_use_list[u].user->kind() == torch::jit::aten::IntImplicit || - in_use_list[u].user->kind() == torch::jit::prim::tolist) { - is_shape_tensor = true; - _in_shape_tensor_index.emplace(i); - break; - } - } - } - // 上面输入为tensor scalar(nv_type是nvinfer1::DataType::kINT32且nv_dims.nbDims <= 1)的状况 - // 是我们自己通过AdjustmentSalarInputs加的,有int_intlist_values_map预热数据支持,可获取到真实的max、min、opt, - // 而不外乎有其他tensor scalar输入的情况,此时由value_dynamic_shape_map记录的max min opt全为0, - // 输入到engine后面converter会报错,这里需要提前拦截。 - // todo: 预热时候给其他tensor scalar加上int_intlist_values_map预热数据支持 - if (nv_dims.nbDims < 1 && !is_shape_tensor) { - LOG(WARNING) << "init_engine_inputs failed: reason: Meet unknown tensor scalar with 0 dim."; - return -1; - } - - nvinfer1::ITensor* trt_in = nullptr; - if (is_shape_tensor) { - c10::List int_sizes = {1}; - nv_dims = nv_dims.nbDims == 0 ? sizes_to_nvdim(int_sizes) : nv_dims; - trt_in = _network->addInput(name.c_str(), nv_type, nv_dims); - int32_t nbvalues = nv_dims.d[0]; - - - std::unique_ptr max_values(new int32_t[nbvalues]); - std::unique_ptr min_values(new int32_t[nbvalues]); - std::unique_ptr opt_values(new int32_t[nbvalues]); - - if (PorosGlobalContext::instance()._value_dynamic_shape_map.count(node_in) == 0) { - LOG(WARNING) << "can't find %" << node_in->debugName() << " in global _value_dynamic_shape_map!"; - return -1; - } - ValueDynamicShape int_value_max_min_opt; - int_value_max_min_opt = PorosGlobalContext::instance()._value_dynamic_shape_map[node_in]; - - std::vector min_values_in_map = int_value_max_min_opt.min_shapes; - std::vector max_values_in_map = int_value_max_min_opt.max_shapes; - std::vector opt_values_in_map = int_value_max_min_opt.opt_shapes; - - if ((size_t)nbvalues != min_values_in_map.size() || - (size_t)nbvalues != max_values_in_map.size() || - (size_t)nbvalues != opt_values_in_map.size()) { - LOG(WARNING) << "input %" << node_in->debugName() << " int or int[] length must match the size of max || min || opt vector!"; - return -1; - } - - for (int i = 0; i < nbvalues; i++) { - max_values[i] = max_values_in_map[i]; - min_values[i] = min_values_in_map[i]; - opt_values[i] = opt_values_in_map[i]; - } - - bool ret_min = profile->setShapeValues(trt_in->getName(), nvinfer1::OptProfileSelector::kMIN, min_values.get(), nbvalues); - bool ret_max = profile->setShapeValues(trt_in->getName(), nvinfer1::OptProfileSelector::kMAX, max_values.get(), nbvalues); - bool ret_opt = profile->setShapeValues(trt_in->getName(), nvinfer1::OptProfileSelector::kOPT, opt_values.get(), nbvalues); - - if (ret_min == false || ret_opt == false || ret_max == false) { - LOG(WARNING) << "setDimensions for value: %" << node_in->debugName() << " failed" - << ", min_shape_info: " << sizes_to_nvdim(min_values_in_map) - << ", opt_shape_info: " << sizes_to_nvdim(opt_values_in_map) - << ", max_shape_info: " << sizes_to_nvdim(max_values_in_map); - return -1; - } - - LOG(INFO) << "Init shape tensor input ok: %" << node_in->debugName() - << ", min_shape_info: " << sizes_to_nvdim(min_values_in_map) - << ", opt_shape_info: " << sizes_to_nvdim(opt_values_in_map) - << ", max_shape_info: " << sizes_to_nvdim(max_values_in_map); - - } else { - if (!current_dynamic) { - trt_in = _network->addInput(name.c_str(), nv_type, nv_dims); - LOG(INFO) << "init static tensor input ok : " << nv_dims; - } else { - if (PorosGlobalContext::instance()._value_dynamic_shape_map.count(node_in) <= 0) { - LOG(WARNING) << "can't generate max min opt input setting for value: %" << node_in->debugName(); - return -1; - } - nvinfer1::Dims dynamic_nv_dims = gen_dynamic_dims(node_in); - trt_in = _network->addInput(name.c_str(), nv_type, dynamic_nv_dims); - std::vector min_shapes = PorosGlobalContext::instance()._value_dynamic_shape_map[node_in].min_shapes; - bool ret_min = profile->setDimensions(trt_in->getName(), nvinfer1::OptProfileSelector::kMIN, sizes_to_nvdim(min_shapes)); - std::vector opt_shapes = PorosGlobalContext::instance()._value_dynamic_shape_map[node_in].opt_shapes; - bool ret_opt = profile->setDimensions(trt_in->getName(), nvinfer1::OptProfileSelector::kOPT, sizes_to_nvdim(opt_shapes)); - std::vector max_shapes = PorosGlobalContext::instance()._value_dynamic_shape_map[node_in].max_shapes; - bool ret_max = profile->setDimensions(trt_in->getName(), nvinfer1::OptProfileSelector::kMAX, sizes_to_nvdim(max_shapes)); - if (ret_min == false || ret_opt == false || ret_max == false) { - LOG(WARNING) << "setDimensions for value: %" << node_in->debugName() << " failed" - << ", min_shape_info: " << sizes_to_nvdim(min_shapes) - << ", opt_shape_info: " << sizes_to_nvdim(opt_shapes) - << ", max_shape_info: " << sizes_to_nvdim(max_shapes) - << ", dynamic tensor info: " << dynamic_nv_dims; - return -1; - } - LOG(INFO) << "Init dynamic tensor input ok: " << nv_dims - << ", min_shape_info: " << sizes_to_nvdim(min_shapes) - << ", opt_shape_info: " << sizes_to_nvdim(opt_shapes) - << ", max_shape_info: " << sizes_to_nvdim(max_shapes); - } - } - _context.set_tensor(in, trt_in); - } - - if (total_is_dynamic) { - POROS_CHECK(profile->isValid(), "Optimization profile is invalid, please check the input range provided"); - _cfg->addOptimizationProfile(profile); - } - return 0; -} - -int TensorrtEngine::mark_graph_outputs(at::ArrayRef outputs) { - int index = 0; - for (const torch::jit::Value* out : outputs) { - auto out_tensor = _context.get_tensor(out); - if (out_tensor == nullptr) { - LOG(WARNING) << "can't get output tensor from context. something is wrong"; - return -1; - } - //output should always be a tensor according to the segmentation setting. - std::string name = std::string("output_") + std::to_string(index++); - out_tensor->setName(name.c_str()); - _network->markOutput(*out_tensor); - LOG(INFO) << "mark " << out->debugName() << " named " << name << " as graph output"; - } - return 0; -} - -//DEPRECATED -std::string TensorrtEngine::convert_graph_to_engine(std::shared_ptr& graph) { - const torch::jit::Block* block = graph->block(); - ConvertersMap* converter_map = PorosGlobalContext::instance().get_converter_map(who_am_i()); - if (converter_map == nullptr) { - LOG(ERROR) << "could not find given engine [ " << who_am_i() << " ] in global context"; - return ""; - } - - for (const torch::jit::Node* node : block->nodes()) { - IConverter* conv = converter_map->get_converter(node); - LOG(INFO) << "start to converter for: " << node_info(node); - if (!conv->converter(this, node)) { - LOG(ERROR) << "converter for node failed [ " << *node->maybeSchema() << " ]"; - return ""; - } - } - - at::ArrayRef outputs = block->outputs(); - if (mark_graph_outputs(outputs) < 0) { - LOG(ERROR) << " mark graph outputs failed"; - return ""; - } - - nvinfer1::ICudaEngine* engine = _builder->buildEngineWithConfig(*_network, *_cfg); - if (!engine) { - LOG(FATAL) << "build tensorrt engine failed"; - } - - nvinfer1::IHostMemory* serialized_engine = engine->serialize(); - engine->destroy(); - std::string engine_str = std::string((const char*)serialized_engine->data(), serialized_engine->size()); - serialized_engine->destroy(); - return engine_str; -} - -void TensorrtEngine::register_module_attribute(const std::string& name, torch::jit::Module& module) { - //auto engine_ptr = c10::make_intrusive(*static_cast(this)); - auto engine_ptr = c10::make_intrusive(*this); - - module.register_attribute( - name, - c10::getCustomClassType>(), - c10::IValue(std::move(engine_ptr)), - false); -} - -std::vector TensorrtEngine::excute_engine(const std::vector& inputs) { - std::vector gpu_handles; - - std::vector contig_inputs{}; - contig_inputs.reserve(inputs.size()); - - for (size_t i = 0; i < inputs.size(); i++) { - uint64_t pyt_idx = _in_binding_map[i]; - //auto expected_type = nvtype_to_attype(_exec_ctx->getEngine().getBindingDataType(i)); - // POROS_CHECK(inputs[pyt_idx].dtype() == expected_type, - // "Expected input tensors to have type " << expected_type << ", found type " << inputs[pyt_idx].dtype()); - - nvinfer1::Dims dims = sizes_to_nvdim_with_pad(inputs[pyt_idx].sizes(), 1); - std::vector shape = nvdim_to_sizes(dims); - // at::ScalarType::Long -> at::ScalarType::Int - if (inputs[pyt_idx].scalar_type() == c10::ScalarType::Long) { - LOG(WARNING) << "excute_engine input meets c10::ScalarType::Long tensor type, change this to c10::ScalarType::Int. " - << "Attention: this may leed to percision change"; - contig_inputs.push_back(inputs[pyt_idx].to(at::ScalarType::Int).view(shape).contiguous()); - } else { - contig_inputs.push_back(inputs[pyt_idx].view(shape).contiguous()); - } - - // 输入可能不在cuda上面,要tocuda - if (contig_inputs[i].device() != c10::DeviceType::CUDA) { - contig_inputs[i] = contig_inputs[i].to(c10::kCUDA).contiguous(); - } - // set input shape binding for nvidia shape tensor - if (_in_shape_tensor_index.count(i) > 0) { - size_t data_nb = inputs[pyt_idx].sizes()[0]; - if (data_nb == 0) { - int32_t set_shape_int = c10::IValue(inputs[pyt_idx].item()).toInt(); - if (!_exec_ctx->setInputShapeBinding(i, &set_shape_int)) { - throw std::runtime_error("tensorrt setInputShapeBinding error"); - } - } else { - std::unique_ptr set_shape_ints(new int32_t[data_nb]); - for (size_t s = 0; s < data_nb; s++) { - c10::IValue tmp_ivalue(inputs[pyt_idx][s].item()); - if (tmp_ivalue.isInt()) { - set_shape_ints[s] = tmp_ivalue.toInt(); - } - } - if (!_exec_ctx->setInputShapeBinding(i, set_shape_ints.get())) { - throw std::runtime_error("tensorrt setInputShapeBinding error"); - } - } - - } else { - if (_exec_ctx->setBindingDimensions(i, dims) == false) { - throw std::runtime_error("tensorrt setBindingDimensions error"); - } - } - gpu_handles.push_back(contig_inputs.back().data_ptr()); - } - - std::vector outputs(_num_io.second); - for (size_t o = inputs.size(); o < (_num_io.first + _num_io.second); o++) { - uint64_t pyt_idx = _out_binding_map[o]; - nvinfer1::Dims out_shape = _exec_ctx->getBindingDimensions(o); - std::vector dims = nvdim_to_sizes(out_shape); - at::ScalarType type = nvtype_to_attype(_exec_ctx->getEngine().getBindingDataType(o)); - outputs[pyt_idx] = std::move(at::empty(dims, {at::kCUDA}).to(type).contiguous()); - gpu_handles.push_back(outputs[pyt_idx].data_ptr()); - } - - c10::cuda::CUDAStream stream = c10::cuda::getCurrentCUDAStream(inputs[0].device().index()); - { - std::lock_guard lock(*_mutex); - _exec_ctx->enqueueV2(gpu_handles.data(), stream, nullptr); - } - return outputs; - -} - -std::vector execute_engine(const std::vector& inputs, - c10::intrusive_ptr compiled_engine) { - return compiled_engine->excute_engine(inputs); -} - -TORCH_LIBRARY(TensorrtEngine, m) { - auto engine_class = m.class_("TensorrtEngine") - .def(torch::init<>()) - .def_pickle( - [](const c10::intrusive_ptr& self) -> std::string { - auto serialized_engine = self->cuda_engine()->serialize(); - return std::string((const char*)serialized_engine->data(), serialized_engine->size()); - }, - [](std::string seralized_engine) -> c10::intrusive_ptr { - return c10::make_intrusive(std::move(seralized_engine)); - }); - m.def("execute_engine", execute_engine); -} - -POROS_REGISTER_ENGINE(TensorrtEngine); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/engine/tensorrt_engine.h b/poros/poros/engine/tensorrt_engine.h deleted file mode 100644 index fbbc031e22d..00000000000 --- a/poros/poros/engine/tensorrt_engine.h +++ /dev/null @@ -1,193 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file tensorrt_engine.h -* @author huangben@baidu.com -* @date Mon Mar 8 11:36:11 CST 2021 -* @brief -**/ - -#pragma once - -//from cuda -#include - -//from pytorch -#include -#include - -//from tensorrt -#include -#include - -#include "poros/compile/poros_module.h" -#include "poros/engine/engine_context.h" -#include "poros/engine/iengine.h" -#include "poros/engine/trtengine_util.h" -#include "poros/log/tensorrt_logging.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/** - * the implement of tensorRT engine - **/ - -class TensorrtEngine : public IEngine { -public: - TensorrtEngine(); - TensorrtEngine(std::string engine_str); - //virtual ~TensorrtEngine(); - - /** - * @brief init - * @return int - * @retval 0 => success, <0 => fail - **/ - virtual int init() override; - - /** - * @brief 核心实现 - * 编译期将subgraph转化成对应engine的图结构保存在engine内部,以使得运行期的excute_engine能调用, 此处保证所有的op都被支持, - * @ 注意要注册输入输出tensor到engine_context中 - * @param [in] sub_graph : 子图 - * @return [res]int - * @retval 0 => success, <0 => fail - **/ - virtual int transform(const PorosGraph& sub_graph) override; - - /** - * @brief 子图执行期逻辑 - * @param [in] inputs : 输入tensor - * @return [res] 输出tensor - **/ - virtual std::vector excute_engine(const std::vector& inputs) override; - - /** - * @brief 在jit 模块中标记engine - * @param [in] name : engine sign - * @param [in] module : jit modeule - * @param [out] module : 添加了engine sign之后的module - **/ - virtual void register_module_attribute(const std::string& name, torch::jit::Module& module) override; - - /** - * @brief get engine mark - * @retval engine name - **/ - virtual const std::string who_am_i() override { - return "TensorrtEngine"; - } - - /** - * @brief get context - * @retval context - **/ - EngineContext& context() { - return _context; - } - - /** - * @brief get network - * @retval network - **/ - nvinfer1::INetworkDefinition* network() { - return _network.get(); - } - - /** - * @brief get cuda engine - * @retval engine - **/ - nvinfer1::ICudaEngine* cuda_engine() { - return _cuda_engine.get(); - } - -private: - - /** - * @brief convert input type from torch to tensorrt - * @param [in] input : input value - * @param [in] input_type : input valur type - **/ - //DEPRECATED - void gen_tensorrt_input_type(const torch::jit::Value* input, - nvinfer1::DataType& input_type); - - /** - * @brief extract input value from subgraph_node - * @param [in] sub_graph : poros graph - * @retval 0 => success, <0 => fail - **/ - int init_engine_inputs(const PorosGraph& sub_graph); - - /** - * @brief mark a tensor as a network output. - * @param [in] outputs : outputs value list - * @retval 0 => success, <0 => fail - **/ - int mark_graph_outputs(at::ArrayRef outputs); - - /** - * @brief binding input and output for engine - **/ - void binding_io(); - - /** - * @brief convert jit graph to engine - * @param [in] graph : jit grpah - * @retval tetengin serialize data - **/ - //DEPRECATED - std::string convert_graph_to_engine(std::shared_ptr& graph); - - /** - * @brief gen dynamic dims for given value - * @param [in] value : jit value - * @retval value dims - **/ - nvinfer1::Dims gen_dynamic_dims(torch::jit::Value* value); - -private: - //for tensortrt networkbuilding - baidu::mirana::poros::TensorrtLogger _logger; - std::shared_ptr _builder; - std::shared_ptr _network; - std::shared_ptr _cfg; - - //engine conrtext. to store the relationship of value-itensor - baidu::mirana::poros::EngineContext _context; - - //for runtime - std::shared_ptr _runtime; - std::shared_ptr _cuda_engine; - std::shared_ptr _exec_ctx; - - std::unordered_map _in_binding_map; - std::unordered_map _out_binding_map; - - std::unordered_set _in_shape_tensor_index; - - PorosOptions _poros_options; - std::shared_ptr _mutex; //for enqueue -}; - -//std::vector execute_engine(const std::vector inputs, -// c10::intrusive_ptr compiled_engine); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/engine/trtengine_util.cpp b/poros/poros/engine/trtengine_util.cpp deleted file mode 100644 index 01fbe006557..00000000000 --- a/poros/poros/engine/trtengine_util.cpp +++ /dev/null @@ -1,364 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file trtengine_util.cpp -* @author tianjinjin@baidu.com -* @date Wed Jul 21 11:45:49 CST 2021 -* @brief -**/ -#include "poros/context/poros_global.h" -#include "poros/engine/trtengine_util.h" -#include "poros/util/poros_util.h" -#include "poros/util/macros.h" - -namespace baidu { -namespace mirana { -namespace poros { - -namespace { -const std::unordered_map& get_at_trt_type_map() { - static const std::unordered_map at_trt_type_map = { - {at::kFloat, nvinfer1::DataType::kFLOAT}, - {at::kHalf, nvinfer1::DataType::kHALF}, - {at::kInt, nvinfer1::DataType::kINT32}, - {at::kChar, nvinfer1::DataType::kINT8}, - {at::kBool, nvinfer1::DataType::kBOOL}, - {at::kByte, nvinfer1::DataType::kINT8}, - }; - return at_trt_type_map; -} - -const std::unordered_map& get_trt_at_type_map() { - static const std::unordered_map trt_at_type_map = { - {nvinfer1::DataType::kFLOAT, at::kFloat}, - {nvinfer1::DataType::kHALF, at::kHalf}, - {nvinfer1::DataType::kINT32, at::kInt}, - {nvinfer1::DataType::kINT8, at::kByte}, //TODO: should trans kChar or kByte??? - {nvinfer1::DataType::kBOOL, at::kBool}, - }; - return trt_at_type_map; -} -} // namespace - -bool broadcastable(nvinfer1::Dims a, nvinfer1::Dims b, bool multidirectional) { - if (a == b) { - return true; - } - - if (multidirectional) { - nvinfer1::Dims a_dims_eq; - nvinfer1::Dims b_dims_eq; - if (a.nbDims > b.nbDims) { - a_dims_eq = a; - b_dims_eq = sizes_to_nvdim_with_pad(nvdim_to_sizes(b), a.nbDims); - } else if (a.nbDims < b.nbDims) { - a_dims_eq = sizes_to_nvdim_with_pad(nvdim_to_sizes(a), b.nbDims); - b_dims_eq = b; - } else { - a_dims_eq = a; - b_dims_eq = b; - } - - bool broadcastable = true; - for (int i = 0; i < a_dims_eq.nbDims; i++) { - if (b_dims_eq.d[i] == a_dims_eq.d[i] || (b_dims_eq.d[i] == 1 || a_dims_eq.d[i] == 1)) { - continue; - } else { - broadcastable = false; - break; - } - } - return broadcastable; - } else { - nvinfer1::Dims b_dims_eq; - if (a.nbDims > b.nbDims) { - b_dims_eq = sizes_to_nvdim_with_pad(nvdim_to_sizes(b), a.nbDims); - } else if (a.nbDims < b.nbDims) { - return false; - } else { - b_dims_eq = b; - } - - bool broadcastable = true; - for (int i = 0; i < a.nbDims; i++) { - if (b_dims_eq.d[i] == a.d[i] || b_dims_eq.d[i] == 1) { - continue; - } else { - broadcastable = false; - break; - } - } - return broadcastable; - } -} - -nvinfer1::Dims sizes_to_nvdim(const std::vector& sizes) { - if (sizes.size() > nvinfer1::Dims::MAX_DIMS) { - LOG(FATAL) << "given sizes is exceed of max dims of tensorrt"; - throw std::runtime_error("given sizes is exceed of max dims of tensorrt"); - } - nvinfer1::Dims dims; - dims.nbDims = sizes.size(); - for (size_t i = 0; i < sizes.size(); i++) { - dims.d[i] = sizes[i]; - } - return dims; -} - -nvinfer1::Dims sizes_to_nvdim(c10::IntArrayRef sizes) { - if (sizes.size() > nvinfer1::Dims::MAX_DIMS) { - LOG(FATAL) << "given sizes is exceed of max dims of tensorrt"; - throw std::runtime_error("given sizes is exceed of max dims of tensorrt"); - } - nvinfer1::Dims dims; - dims.nbDims = sizes.size(); - for (size_t i = 0; i < sizes.size(); i++) { - dims.d[i] = sizes[i]; - } - return dims; -} - -nvinfer1::Dims sizes_to_nvdim(c10::List sizes) { - if (sizes.size() > nvinfer1::Dims::MAX_DIMS) { - LOG(FATAL) << "given sizes is exceed of max dims of tensorrt"; - throw std::runtime_error("given sizes is exceed of max dims of tensorrt"); - } - nvinfer1::Dims dims; - dims.nbDims = sizes.size(); - for (size_t i = 0; i < sizes.size(); i++) { - dims.d[i] = sizes[i]; - } - return dims; -} - -nvinfer1::Dims sizes_to_nvdim_with_pad(c10::IntArrayRef sizes, uint64_t pad_to) { - if (pad_to > nvinfer1::Dims::MAX_DIMS || sizes.size() > nvinfer1::Dims::MAX_DIMS) { - LOG(FATAL) << "given sizes is exceed of max dims of tensorrt"; - throw std::runtime_error("given sizes is exceed of max dims of tensorrt"); - } - - nvinfer1::Dims dims; - //no need padding situation - if (sizes.size() > pad_to) { - dims.nbDims = sizes.size(); - for (size_t i = 0; i < sizes.size(); i++) { - dims.d[i] = sizes[i]; - } - //need padding situation - } else { - dims.nbDims = pad_to; - for (size_t i = 0; i < pad_to - sizes.size(); i++) { - dims.d[i] = 1; - } - for (size_t i = pad_to - sizes.size(); i < pad_to; i++) { - dims.d[i] = sizes[i - (pad_to - sizes.size())]; - } - } - return dims; -} - -nvinfer1::Dims sizes_to_nvdim_with_pad(c10::List sizes, uint64_t pad_to) { - if (pad_to > nvinfer1::Dims::MAX_DIMS || sizes.size() > nvinfer1::Dims::MAX_DIMS) { - LOG(FATAL) << "given sizes is exceed of max dims of tensorrt"; - throw std::runtime_error("given sizes is exceed of max dims of tensorrt"); - } - - nvinfer1::Dims dims; - //no need padding situation - if (sizes.size() > pad_to) { - LOG(INFO) << "no need to pad, give sizes: " << sizes.size() - << ", expected dims: " << pad_to; - dims.nbDims = sizes.size(); - for (size_t i = 0; i < sizes.size(); i++) { - dims.d[i] = sizes[i]; - } - //need padding situation - } else { - dims.nbDims = pad_to; - for (size_t i = 0; i < pad_to - sizes.size(); i++) { - dims.d[i] = 1; - } - for (size_t i = pad_to - sizes.size(); i < pad_to; i++) { - dims.d[i] = sizes[i - (pad_to - sizes.size())]; - } - } - return dims; -} - -std::vector nvdim_to_sizes(const nvinfer1::Dims& dim) { - std::vector sizes; - for (int i = 0; i < dim.nbDims; i++) { - sizes.push_back(dim.d[i]); - } - return std::move(sizes); -} - -std::string nvdim_to_str(const nvinfer1::Dims& dim) { - std::stringstream ss; - ss << dim; - return ss.str(); -} - -int64_t nvdim_to_volume(const nvinfer1::Dims& dim) { - return std::accumulate(dim.d, dim.d + dim.nbDims, 1, std::multiplies()); -} - -nvinfer1::Dims unpad_nvdim(const nvinfer1::Dims& dim) { - nvinfer1::Dims new_dim; - int j = 0; - bool pad_dims_done = false; - - for (int i = 0; i < dim.nbDims; i++) { - if (dim.d[i] == 1 && !pad_dims_done) { - // skip over unecessary dimension - continue; - } else { - new_dim.d[j] = dim.d[i]; - j++; - // keep all other dimensions (don't skip over them) - pad_dims_done = true; - } - } - new_dim.nbDims = j; - return new_dim; -} - -bool gen_tensor_type(const torch::jit::Node& node, const size_t index, nvinfer1::DataType& nv_type) { - c10::optional maybe_type; - //at::ArrayRef inputs = node.inputs(); - std::shared_ptr subgraph = node.g(torch::jit::attr::Subgraph); - at::ArrayRef inputs = subgraph->inputs(); - //for (size_t index = 0; index < inputs.size(); index++) { - auto value = inputs[index]; - - //extract scalar type from tensor. - if (value->type()->isSubtypeOf(c10::TensorType::get())) { - c10::TensorTypePtr op = value->type()->cast(); - if (op->scalarType().has_value()) { - maybe_type = op->scalarType().value(); - } - - //extract scalar type from tensorlist. - } else if (value->type()->isSubtypeOf(c10::ListType::ofTensors())) { - auto list_element = value->type()->cast()->getElementType(); - //TODO: ADD SOPPORT HERE - LOG(WARNING) << "gen_tensor_type for tensorlist to add more"; - return false; - } - - //this is added because tensorrt only support five kinds of date type - //(kFloat / kHalf / kINT8 / kINT32 / kBOOL) for now. (2021.08.01) - if (maybe_type.has_value() && maybe_type.value() != at::kFloat && - maybe_type.value() != at::kHalf && maybe_type.value() != at::kChar && - maybe_type.value() != at::kInt && maybe_type.value() != at::kBool) { - // when we meet at::KLong and globalContext allow us to down to at::KInt - if (maybe_type.value() == at::kLong && PorosGlobalContext::instance().get_poros_options().long_to_int == true) { - nv_type = attype_to_nvtype(at::kInt); - LOG(WARNING) << "gen_tensor_type meets at::KLong tensor type, change this to at::KInt. " - << "Attention: this may leed to percision change"; - return true; - } - LOG(WARNING) << "gen_tensor_type failed, reason: " - << "given scalartype is not supported by tensorrt"; - return false; - } - - if (maybe_type.has_value()) { - nv_type = attype_to_nvtype(maybe_type.value()); - return true; - } else { - LOG(WARNING) << "gen_tensor_type failed, reason: " - << "cant't extract scalar type from all the input value"; - return false; - } -} - -at::ScalarType nvtype_to_attype(nvinfer1::DataType type) { - auto trt_at_type_map = get_trt_at_type_map(); - if (trt_at_type_map.find(type) == trt_at_type_map.end()) { - LOG(FATAL) << "unsupported tensorrt datatype"; - throw std::runtime_error("unsupported tensorrt datatype"); - } - return trt_at_type_map.at(type); -} - -nvinfer1::DataType attype_to_nvtype(at::ScalarType type) { - auto at_trt_type_map = get_at_trt_type_map(); - if (at_trt_type_map.find(type) == at_trt_type_map.end()) { - LOG(FATAL) << "unsupported aten datatype"; - throw std::runtime_error("unsupported aten datatype"); - } - return at_trt_type_map.at(type); -} - -nvinfer1::Dims unsqueeze_dims(const nvinfer1::Dims& d, int pos, int val, bool use_zeros) { - // acceptable range for pos is [0, d.nbDims] - POROS_CHECK(pos >= 0 && pos <= d.nbDims, "ERROR: Index to unsqueeze is out of bounds."); - nvinfer1::Dims dims; - for (int i = 0, j = 0; j <= d.nbDims; j++) { - // add new dimension at pos - if (j == pos) { - dims.d[j] = val; - } else { - dims.d[j] = (use_zeros && d.d[i] == -1) ? 0 : d.d[i]; - ++i; - } - } - dims.nbDims = d.nbDims + 1; - return dims; -} - -nvinfer1::Dims squeeze_dims(const nvinfer1::Dims& d, int pos, bool use_zeros) { - // acceptable range for pos is [0, d.nbDims] - POROS_CHECK(pos >= 0 && pos <= d.nbDims, "ERROR: Index to unsqueeze is out of bounds."); - nvinfer1::Dims dims; - int j = 0; - for (int i = 0; i < d.nbDims; i++) { - if (i != pos) { - dims.d[j++] = (use_zeros && d.d[i] == -1) ? 0 : d.d[i]; - } - } - dims.nbDims = j; - return dims; -} - -bool check_nvtensor_is_dynamic(const nvinfer1::ITensor* nvtensor) { - POROS_CHECK(nvtensor != nullptr, "input nvtensor is null"); - nvinfer1::Dims nvtensor_dims = nvtensor->getDimensions(); - for (int i = 0; i < nvtensor_dims.nbDims; i++) { - if (nvtensor_dims.d[i] < 0) { - return true; - } - } - return false; -} - -bool input_is_dynamic(torch::jit::Value* input) { - auto _value_dynamic_shape_map = PorosGlobalContext::instance()._value_dynamic_shape_map; - if (_value_dynamic_shape_map.find(input) != _value_dynamic_shape_map.end()) { - auto min_shapes = _value_dynamic_shape_map[input].min_shapes; - auto max_shapes = _value_dynamic_shape_map[input].max_shapes; - for(size_t i = 0; i < min_shapes.size(); i++) { - if (max_shapes[i] != min_shapes[i]) { - return true; - } - } - } - return false; -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/engine/trtengine_util.h b/poros/poros/engine/trtengine_util.h deleted file mode 100644 index 93aa6297089..00000000000 --- a/poros/poros/engine/trtengine_util.h +++ /dev/null @@ -1,130 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file trtengine_util.h -* @author tianjinjin@baidu.com -* @date Wed Jul 21 11:45:49 CST 2021 -* @brief -**/ - -#pragma once - -//from pytorch -#include "torch/script.h" -//from tensorrt -#include "NvInfer.h" - -namespace baidu { -namespace mirana { -namespace poros { - -//实现nvinfer::Dims的 == 运算符 -inline bool operator==(const nvinfer1::Dims& in1, const nvinfer1::Dims& in2) { - if (in1.nbDims != in2.nbDims) { - return false; - } - // TODO maybe look to support broadcasting comparisons - for (int64_t i = 0; i < in1.nbDims; i++) { - if (in1.d[i] != in2.d[i]) { - return false; - } - } - return true; -} - -//实现nvinfer::Dims的 != 运算符 -inline bool operator!=(const nvinfer1::Dims& in1, const nvinfer1::Dims& in2) { - return !(in1 == in2); -} - -//实现nvinfer::Dims的<<运算符 -template -inline std::ostream& print_sequence(std::ostream& stream, const T* begin, int count) { - stream << "["; - if (count > 0) { - std::copy_n(begin, count - 1, std::ostream_iterator(stream, ", ")); - stream << begin[count - 1]; - } - stream << "]"; - return stream; -} - -inline std::ostream& operator<<(std::ostream& stream, const nvinfer1::Dims& shape) { - return print_sequence(stream, shape.d, shape.nbDims); -} - -//实现nvinfer::DataType的<<运算符 -inline std::ostream& operator<<(std::ostream& stream, const nvinfer1::DataType& dtype) { - switch (dtype) { - case nvinfer1::DataType::kFLOAT: - return stream << "Float32"; - case nvinfer1::DataType::kHALF: - return stream << "Float16"; - case nvinfer1::DataType::kINT8: - return stream << "Int8"; - case nvinfer1::DataType::kINT32: - return stream << "Int32"; - case nvinfer1::DataType::kBOOL: - return stream << "Bool"; - default: - return stream << "Unknown Data Type"; - } -} - -// 创建智能指针 -template -std::shared_ptr make_shared_ptr(T* p) { - return std::shared_ptr(p); -} - -//int64_t volume(const nvinfer1::Dims& dim); //move to nvdim_to_volume -bool broadcastable(nvinfer1::Dims a, nvinfer1::Dims b, bool multidirectional = true); - -//以下四个函数,实现tensorrt的dims结构与vec形式的sizes的互换。 -nvinfer1::Dims sizes_to_nvdim(const std::vector& sizes); -nvinfer1::Dims sizes_to_nvdim(c10::IntArrayRef sizes); -nvinfer1::Dims sizes_to_nvdim(c10::List sizes); -nvinfer1::Dims sizes_to_nvdim_with_pad(c10::IntArrayRef sizes, uint64_t pad_to); -nvinfer1::Dims sizes_to_nvdim_with_pad(c10::List sizes, uint64_t pad_to); -//以下三个函数,实现tensorrt的dim到其他形式的转换。 -std::vector nvdim_to_sizes(const nvinfer1::Dims& dim); -std::string nvdim_to_str(const nvinfer1::Dims& dim); -int64_t nvdim_to_volume(const nvinfer1::Dims& dim); - -//以下一个函数,实现dim的unpad -nvinfer1::Dims unpad_nvdim(const nvinfer1::Dims& dim); - -//以下两个函数,实现tensorrt与aten的类型互转 -//transform tensorrt-type to aten-type(which used in pytorch and torchscript) -at::ScalarType nvtype_to_attype(nvinfer1::DataType type); -//transform aten-type(which used in pytorch and torchscript) to tensorrt-type -nvinfer1::DataType attype_to_nvtype(at::ScalarType type); - -//以下两个函数,实现tensorrt的dims的展开和压缩(???)。 -nvinfer1::Dims unsqueeze_dims(const nvinfer1::Dims& d, int pos, int val = 1, bool use_zeros = true); -nvinfer1::Dims squeeze_dims(const nvinfer1::Dims& d, int pos, bool use_zeros = true); - -/* -* @brief 通过node的输入信息,获取相应的tensor的类型。 -**/ -bool gen_tensor_type(const torch::jit::Node& node, const size_t index, nvinfer1::DataType & nv_type); -// 检查输入nvtensor是否为dynamic输入 -bool check_nvtensor_is_dynamic(const nvinfer1::ITensor* nvtensor); -// 检查一个子图输入是否为dynamic -bool input_is_dynamic(torch::jit::Value* input); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/iplugin/plugin_create.cpp b/poros/poros/iplugin/plugin_create.cpp deleted file mode 100644 index 21e0d9e2e4a..00000000000 --- a/poros/poros/iplugin/plugin_create.cpp +++ /dev/null @@ -1,91 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - - - /** - * @file plugin_create.cpp - * @author huangben(huangben@baidu.com) - * @date 2018/10/23 14:16:18 - * @version $Revision$ - * @brief - **/ -#include "poros/iplugin/plugin_create.h" -#include "poros/log/poros_logging.h" - -namespace baidu { -namespace mirana { -namespace poros { - - static plugin_creator_map_t _g_creator_map; - - void register_plugin_creator(const std::string& plugin_name, plugin_creator_t creator) { - if (_g_creator_map.find(plugin_name) != _g_creator_map.end()) { - //throw bsl::KeyAlreadyExistException() << BSL_EARG - // << "[plugin_name:" << plugin_name << "]"; - LOG(ERROR) << plugin_name << " had resiterd! there have more than 1 plugin use samename"; - } - _g_creator_map[plugin_name] = creator; - } - - void register_plugin_creator(const std::string& plugin_name, plugin_creator_t creator, - plugin_creator_map_t& plugin_creator_map) { - - if (plugin_creator_map.find(plugin_name) != plugin_creator_map.end()) { - //throw bsl::KeyAlreadyExistException() << BSL_EARG - // << "[plugin_name:" << plugin_name << "]"; - LOG(ERROR) << plugin_name << " had resiterd! there have more than 1 plugin use samename"; - } - plugin_creator_map[plugin_name] = creator; - } - - IPlugin* create_plugin(const std::string& plugin_name) { - plugin_creator_map_t::const_iterator it; - - it = _g_creator_map.find(plugin_name); - if (it == _g_creator_map.end()) { - LOG(FATAL) << "No such plugin type:" << plugin_name; - return NULL; - } - return it->second(); - } - - IPlugin* create_plugin(const std::string& plugin_name, const plugin_creator_map_t& plugin_creator_map) { - plugin_creator_map_t::const_iterator it; - - it = plugin_creator_map.find(plugin_name); - if (it == plugin_creator_map.end()) { - LOG(FATAL) << "No such plugin type:" << plugin_name; - return NULL; - } - return it->second(); - } - - //void create_all_plugins(std::unordered_map& plugin_m) { - // for (auto& e : _g_creator_map) { - // plugin_m[e.first] = e.second(); - // } - //} - void create_all_plugins(const plugin_creator_map_t& plugin_creator_map, - std::unordered_map& plugin_m) { - for (auto& e : plugin_creator_map) { - plugin_m[e.first] = e.second(); - } - } - -}//poros -}//mirana -}//baidu - - -/* vim: set ts=4 sw=4 sts=4 tw=100 */ diff --git a/poros/poros/iplugin/plugin_create.h b/poros/poros/iplugin/plugin_create.h deleted file mode 100644 index 586a2a0468b..00000000000 --- a/poros/poros/iplugin/plugin_create.h +++ /dev/null @@ -1,73 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - - - /** - * @file plugin_create.h - * @author huangben(huangben@baidu.com) - * @date 2018/10/23 14:16:18 - * @version $Revision$ - * @brief - **/ -#pragma once - -#include -#include - -namespace baidu { -namespace mirana { -namespace poros { - -class IPlugin { -public: - virtual ~IPlugin() {} - virtual const std::string who_am_i() = 0; -}; - -typedef IPlugin* (*plugin_creator_t)(); -typedef std::unordered_map plugin_creator_map_t; - -IPlugin* create_plugin(const std::string& plugin_name); -IPlugin* create_plugin(const std::string& plugin_name, const plugin_creator_map_t& plugin_creator_map); - -void create_all_plugins(const plugin_creator_map_t& plugin_creator_map, - std::unordered_map& plugin_m); -//void create_all_plugins(std::unordered_map& plugin_m); - -template -IPlugin* default_plugin_creator() { - return new (std::nothrow)PluginType; -} - -void register_plugin_creator(const std::string& plugin_name, plugin_creator_t creator); -void register_plugin_creator(const std::string& plugin_name, - plugin_creator_t creator, plugin_creator_map_t& plugin_creator_map); - -template -void register_plugin_class(const std::string& plugin_name) { - return register_plugin_creator(plugin_name, default_plugin_creator); -} - -//推荐使用此版本 -template -void register_plugin_class(const std::string& plugin_name, plugin_creator_map_t& plugin_creator_map) { - return register_plugin_creator(plugin_name, default_plugin_creator, plugin_creator_map); -} - -}//poros -}//mirana -}//baidu - - -/* vim: set ts=4 sw=4 sts=4 tw=100 */ diff --git a/poros/poros/log/poros_logging.h b/poros/poros/log/poros_logging.h deleted file mode 100644 index 0cbe1d141d1..00000000000 --- a/poros/poros/log/poros_logging.h +++ /dev/null @@ -1,25 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file poros_logging.h -* @author tianjinjin@baidu.com -* @date Wed Jun 2 20:54:24 CST 2021 -* @brief -**/ - -#pragma once - -//from pytorch -#include "c10/util/Logging.h" diff --git a/poros/poros/log/tensorrt_logging.cpp b/poros/poros/log/tensorrt_logging.cpp deleted file mode 100644 index 65064b10939..00000000000 --- a/poros/poros/log/tensorrt_logging.cpp +++ /dev/null @@ -1,97 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file tensorrt_logging.cpp -* @author tianjinjin@baidu.com -* @date Wed Jun 2 21:14:23 CST 2021 -* @brief -**/ - -#include -#include "poros/context/poros_global.h" -#include "poros/log/tensorrt_logging.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/* -TensorrtLogger::TensorrtLogger(logging::LogSeverity severity) { - _severity = severity; - switch (severity) { - case logging::BLOG_INFO: - _nv_level = nvinfer1::ILogger::Severity::kINFO; - break; - case logging::BLOG_NOTICE: - _nv_level = nvinfer1::ILogger::Severity::kVERBOSE; - break; - case logging::BLOG_WARNING: - _nv_level = nvinfer1::ILogger::Severity::kWARNING; - break; - case logging::BLOG_ERROR: - _nv_level = nvinfer1::ILogger::Severity::kERROR; - break; - case logging::BLOG_FATAL: - _nv_level = nvinfer1::ILogger::Severity::kINTERNAL_ERROR; - break; - default: - break; - } -} */ - -TensorrtLogger::TensorrtLogger() { - auto debug = PorosGlobalContext::instance().get_poros_options().debug; - _torch_level = debug ? 0 : 1; - _nv_level = debug ? nvinfer1::ILogger::Severity::kVERBOSE : - nvinfer1::ILogger::Severity::kWARNING; -} - -TensorrtLogger::TensorrtLogger(uint32_t torch_level) { - _torch_level = torch_level; - switch (torch_level) { - case 3: /*c10::GLOG_FATAL*/ - _nv_level = nvinfer1::ILogger::Severity::kINTERNAL_ERROR; - break; - case 2: /*c10::GLOG_ERROR*/ - _nv_level = nvinfer1::ILogger::Severity::kERROR; - break; - case 1: /*c10::GLOG_WARNING*/ - _nv_level = nvinfer1::ILogger::Severity::kWARNING; - break; - case 0: /*c10::GLOG_INFO*/ - _nv_level = nvinfer1::ILogger::Severity::kVERBOSE; - break; - default: /*c10::GLOG_WARNING*/ - _nv_level = nvinfer1::ILogger::Severity::kWARNING; - break; - } -} - -void TensorrtLogger::log(nvinfer1::ILogger::Severity severity, const char* msg) noexcept { - // suppress unprintable messages - if (severity > _nv_level) { - return; - } - std::cout << msg << std::endl; //TO MAKE THIS BETTER. -} - -TensorrtLogger& get_nvlogger() { - static TensorrtLogger nv_logger; - return nv_logger; -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/log/tensorrt_logging.h b/poros/poros/log/tensorrt_logging.h deleted file mode 100644 index 795f27e3ef0..00000000000 --- a/poros/poros/log/tensorrt_logging.h +++ /dev/null @@ -1,59 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file tensorrt_logging.h -* @author tianjinjin@baidu.com -* @date Wed Jun 2 20:54:24 CST 2021 -* @brief -**/ - -#pragma once - -#include - -//from pytorch -#include "c10/util/Logging.h" - -//from tensorrt -#include "NvInfer.h" - -#include "poros/log/poros_logging.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/** - * the required logger setting for tensorrt engine - * **/ -class TensorrtLogger : public nvinfer1::ILogger { -public: - TensorrtLogger(); - TensorrtLogger(uint32_t torch_level); - void log(nvinfer1::ILogger::Severity severity, const char* msg) noexcept; - uint32_t torch_level() { - return _torch_level; - } - -private: - uint32_t _torch_level = 1; - nvinfer1::ILogger::Severity _nv_level; -}; - -TensorrtLogger& get_nvlogger(); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/lowering/eliminate_exception_pass.cpp b/poros/poros/lowering/eliminate_exception_pass.cpp deleted file mode 100644 index 133e519483c..00000000000 --- a/poros/poros/lowering/eliminate_exception_pass.cpp +++ /dev/null @@ -1,119 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file eliminate_exception_pass.cpp -* @author tianjinjin@baidu.com -* @date Thu Sep 23 11:15:49 CST 2021 -* @brief -**/ -#include "poros/lowering/lowering_pass.h" - -#include - -namespace baidu { -namespace mirana { -namespace poros { - -namespace { -using namespace torch::jit; -struct EliminateExceptionPasses { - EliminateExceptionPasses(std::shared_ptr graph) : graph_(std::move(graph)) {} - - void run() { - find_exception_if_node(graph_->block()); - torch::jit::EliminateDeadCode(graph_); - } - -private: - bool is_exception_if_node(Node* n) { - /// Check if this Node hosts a pattern like so: - /// situation 1: - /// = prim::If(%5958) - /// block0(): - /// -> () - /// block1(): - /// = prim::RaiseException(%45) - /// -> () - - /// situation 2: - /// = prim::If(%5958) - /// block0(): - /// = prim::RaiseException(%45) - /// -> () - /// block1(): - /// -> () - if (n->blocks().size() != 2) { - return false; - } - auto arm1 = n->blocks()[0]; - auto arm2 = n->blocks()[1]; - if (arm1->outputs().size() != 0 || arm2->outputs().size() != 0) { - // Make sure that the node doesn't actually produce any Value that are - // used by other nodes - return false; - } - - auto arm1_start = arm1->nodes().begin(); - auto arm2_start = arm2->nodes().begin(); - - if ((*arm1_start)->kind() == prim::Return) { - // Make sure that block0 is solely the return - if ((*arm2_start)->kind() != prim::RaiseException || (*(++arm2_start))->kind() != prim::Return) { - // Make sure that block1 is solely just the exception and the return - return false; - } - return true; - } - - if ((*arm2_start)->kind() == prim::Return) { - // Make sure that block1 is solely the return - if ((*arm1_start)->kind() != prim::RaiseException || (*(++arm1_start))->kind() != prim::Return) { - // Make sure that block0 is solely just the exception and the return - return false; - } - return true; - } - return false; - } - - void find_exception_if_node(Block* b) { - for (auto it = b->nodes().begin(); it != b->nodes().end(); it++) { - auto n = *it; - if (n->kind() == prim::If && is_exception_if_node(n)) { - it.destroyCurrent(); - } else if (n->kind() == prim::If) { - auto true_block = n->blocks()[0]; - find_exception_if_node(true_block); - auto false_block = n->blocks()[1]; - find_exception_if_node(false_block); - } else if (n->kind() == prim::Loop) { - auto loop_block = n->blocks()[0]; - find_exception_if_node(loop_block); - } - } - } - - std::shared_ptr graph_; -}; -} // namespace - -void eliminate_exception_pass(std::shared_ptr graph) { - EliminateExceptionPasses eppe(std::move(graph)); - eppe.run(); -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/lowering/eliminate_maxpoll_with_indices.cpp b/poros/poros/lowering/eliminate_maxpoll_with_indices.cpp deleted file mode 100644 index c18b40208f9..00000000000 --- a/poros/poros/lowering/eliminate_maxpoll_with_indices.cpp +++ /dev/null @@ -1,129 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file eliminate_maxpoll_with_indices.cpp -* @author tianjinjin@baidu.com -* @date Tue Sep 13 11:06:07 CST 2022 -* @brief -**/ -#include "poros/lowering/lowering_pass.h" - -#include -#include -#include -#include -#include - -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -namespace { -using namespace torch::jit; -/** - * @brief 尝试用maxpool 代替 maxpool_with_indeces. - * 以 maxpoll2d 为例: - * maxpoll2d_with_indices 的schema为:aten::max_pool2d_with_indices(Tensor self, int[2] kernel_size, int[2] stride=[], int[2] padding=0, int[2] dilation=1, bool ceil_mode=False) -> (Tensor, Tensor) - * 而 maxpoll 的schema为:aten::max_pool2d(Tensor self, int[2] kernel_size, int[2] stride=[], int[2] padding=0, int[2] dilation=1, bool ceil_mode=False) -> Tensor - * 这两个op,输入参数完全一致,输出上,max_pool2d_with_indices有两个输出,第一个输出与max_pool2d的输出完全一致,第二个输出为indeces信息。 - * 当 max_pool2d_with_indices 的第二个输出indices,后续没有其他op使用该value的时候, - * 我们直接用max_pool2d 替代 max_pool2d_with_indices。 - **/ -struct EliminateMaxpollWithIndices { - EliminateMaxpollWithIndices(std::shared_ptr graph) : graph_(std::move(graph)) {} - - void run() { - GRAPH_DUMP("before eliminate_maxpool_with_indices Graph: ", graph_); - bool changed = eliminate_maxpool_with_indices(graph_->block()); - if (changed) { - ConstantPropagation(graph_); - EliminateDeadCode(graph_); - EliminateCommonSubexpression(graph_); - ConstantPooling(graph_); - } - GRAPH_DUMP("after eliminate_maxpool_with_indices Graph: ", graph_); - return; - } - -private: - - bool is_maxpoll_with_indices_pattern(Node* node) { - if (node->kind() != aten::max_pool1d_with_indices && - node->kind() != aten::max_pool2d_with_indices && - node->kind() != aten::max_pool3d_with_indices){ - return false; - } - - //当outputs的第二个值,也就是indices,没有被其他op使用的时候,满足替换条件。 - Value* indices = node->output(1); - if (indices->uses().size() == 0) { - return true; - } - return false; - } - - bool replace_maxpool_with_indices(Node* node) { - NodeKind replace_kind = aten::max_pool1d; - if (node->kind() == aten::max_pool2d_with_indices) { - replace_kind = aten::max_pool2d; - } else if (node->kind() == aten::max_pool3d_with_indices) { - replace_kind = aten::max_pool3d; - }; - - Node* maxpool_node = graph_->create(replace_kind, node->inputs()); - maxpool_node->output(0)->setType(node->output(0)->type()); - maxpool_node->copyMetadata(node); - maxpool_node->insertBefore(node); - node->output(0)->replaceAllUsesAfterNodeWith(node, maxpool_node->output(0)); - //node->output(0)->replaceAllUsesWith(maxpool_node->output(0)); - - LOG(INFO) << "destroy maxpool_with_indeces node now: " << node_info(node); - node->destroy(); - return true; - } - - bool eliminate_maxpool_with_indices(Block* block) { - bool changed = false; - for (auto it = block->nodes().begin(); it != block->nodes().end();) { - // we might destroy the current node, so we need to pre-increment - // the iterator - Node* node = *it; - ++it; - for (Block* subblock : node->blocks()) { - changed |= eliminate_maxpool_with_indices(subblock); - } - if (is_maxpoll_with_indices_pattern(node)) { - LOG(INFO) << "find maxpoll with indices pattern: " << node_info(node); - changed |= replace_maxpool_with_indices(node); - } - } - return changed; - } - -std::shared_ptr graph_; -}; - -} // namespace - -void eliminate_maxpool_with_indices(std::shared_ptr graph) { - EliminateMaxpollWithIndices emwi(std::move(graph)); - emwi.run(); -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/lowering/eliminate_simple_useless_nodes.cpp b/poros/poros/lowering/eliminate_simple_useless_nodes.cpp deleted file mode 100644 index 139bcbeb5cf..00000000000 --- a/poros/poros/lowering/eliminate_simple_useless_nodes.cpp +++ /dev/null @@ -1,99 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file eliminate_simple_useless_nodes.cpp -* @author tianshaoqing@baidu.com -* @date 2022-08-25 11:06:26 -* @brief -**/ -#include "poros/lowering/lowering_pass.h" - -#include -#include -#include -#include -#include - -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -namespace { -using namespace torch::jit; - -struct EliminateSimpleUselessNodes { - EliminateSimpleUselessNodes(std::shared_ptr graph) : graph_(std::move(graph)) { - useless_schema_set_.emplace(torch::jit::parseSchema("aten::dropout(Tensor input, float p, " - "bool train) -> Tensor").operator_name()); - useless_schema_set_.emplace(torch::jit::parseSchema("aten::warn(str message, int stacklevel=2) " - "-> ()").operator_name()); - } - - void run() { - GRAPH_DUMP("before eliminate_simple_useless_nodes Graph: ", graph_); - bool changed = find_and_eliminate_simple_useless_nodes(graph_->block()); - if (changed) { - ConstantPropagation(graph_); - EliminateDeadCode(graph_); - EliminateCommonSubexpression(graph_); - ConstantPooling(graph_); - } - GRAPH_DUMP("after eliminate_simple_useless_nodes Graph: ", graph_); - return; - } - -private: - bool find_and_eliminate_simple_useless_nodes(Block* block) { - bool graph_changed = false; - auto it = block->nodes().begin(); - while (it != block->nodes().end()) { - auto node = *it; - ++it; //++it first, node may be destroyed later。 - for (auto sub_block: node->blocks()) { - if (find_and_eliminate_simple_useless_nodes(sub_block)) { - graph_changed = true; - } - } - - if (node->maybeSchema() && useless_schema_set_.count(node->schema().operator_name())) { - if (node->kind() == torch::jit::aten::warn) { - node->destroy(); - } - if (node->kind() == torch::jit::aten::dropout) { - node->output(0)->replaceAllUsesWith(node->input(0)); - node->destroy(); - } - graph_changed = true; - } - } - return graph_changed; - } - - std::shared_ptr graph_; - std::unordered_set useless_schema_set_; -}; - -} // namespace - -void eliminate_simple_useless_nodes(std::shared_ptr graph) { - EliminateSimpleUselessNodes esun(std::move(graph)); - esun.run(); -} - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/lowering/eliminate_some_dict.cpp b/poros/poros/lowering/eliminate_some_dict.cpp deleted file mode 100644 index e43daca5fe0..00000000000 --- a/poros/poros/lowering/eliminate_some_dict.cpp +++ /dev/null @@ -1,166 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file eliminate_some_dict.cpp -* @author tianjinjin@baidu.com -* @date Wed Jan 26 19:41:32 CST 2022 -* @brief -**/ -#include "poros/lowering/lowering_pass.h" - -#include -#include -#include -#include -#include - -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -namespace { -using namespace torch::jit; - -struct EliminateSomeDict { - EliminateSomeDict(std::shared_ptr graph) : graph_(std::move(graph)) {} - - void run() { - GRAPH_DUMP("before eliminate_some_dicts Graph: ", graph_); - bool changed = eliminate_dict_getitems(graph_->block()); - if (changed) { - ConstantPropagation(graph_); - EliminateDeadCode(graph_); - EliminateCommonSubexpression(graph_); - ConstantPooling(graph_); - } - GRAPH_DUMP("after eliminate_some_dicts Graph: ", graph_); - return; - } - -private: - /* - * @brief - * 将以下graph: - * %key1 : str = prim::Constant[value="first_key"]() - * %key2 : str = prim::Constant[value="second_key"]() - * %key3 : str = prim::Constant[value="third_key"]() - * %resdict = prim::DictConstruct(%key1, %value1, %key2, %value2, %key3, %value3) - * %res1 = aten::__getitem__(%resdict, %key1) - * %res2 = aten::__getitem__(%resdict, %key2) - * %1 = aten::matmul(%res1, %const.1) - * %2 = aten::matmul(%res2, %const.2) - * - * 替换成以下graph: - * %1 = aten::matmul(%value1, %const.1) - * %2 = aten::matmul(%value2, %const.2) - * - * 需要特别注意的是: - * 1. 如果有多个__getitem__的时候,prim::DictConstruct不可删除, - * 当所有的__getitem__都处理完了才能删除prim::DictConstruct - * */ - bool is_dict_getitem_node(Node* node) { - if (node->kind() != aten::__getitem__ || - node->inputs().at(1)->node()->kind() != prim::Constant) { - return false; - } - auto producer_node = node->inputs().at(0)->node(); - if (producer_node->kind() != prim::DictConstruct || - producer_node->owningBlock() != node->owningBlock()) { - return false; - } - - //TODO: this can be changed to more loose condition - for (auto &use: node->inputs().at(0)->uses()) { - if (use.user->kind() != aten::__getitem__) { - return false; - } - } - - return true; - } - - bool eliminate_dict_getitem_after_construct(Node* node) { - //提取get_item节点的key信息,即第二个参数的值。 - c10::optional maybe_key = toIValue(node->inputs().at(1)->node()->output()); - if (!maybe_key.has_value()) { - LOG(INFO) << "can not handle get_item node: " << node_info(node); - return false; - } - auto key = maybe_key.value(); - - //找到DictConstruct相应的key, 替换成相应的value。 - auto producer_node = node->inputs().at(0)->node(); - at::ArrayRef producer_inputs = producer_node->inputs(); - size_t num_inputs = producer_inputs.size(); - for(size_t index = 0; index < num_inputs / 2; index++) { - if (producer_inputs[index * 2]->node()->kind() != prim::Constant) { - continue; - // LOG(INFO) << "can not handle DictConstruct node: " << node_info(producer_node); - // return false; - } else { - c10::optional ivalue = toIValue(producer_inputs[index * 2]->node()->output()); - if (ivalue.has_value() && ivalue.value() == key) { - //开启output value 替换大法。 - node->outputs()[0]->replaceAllUsesWith(producer_inputs[index * 2 + 1]); - //本node可以destroy了。 - LOG(INFO) << "replace all uses from value: %" << node->outputs()[0]->debugName() - << " to value: %" << producer_inputs[index * 2 + 1]->debugName(); - LOG(INFO) << "destroy getitem node now: " << node_info(node); - node->destroy(); - break; - } - } - } - - //当只有一个 getitem 节点的时候,producer 可以destroy了。(无需专门删除,EliminateDeadCode会处理掉producer_node) - // if (producer_node->outputs()[0]->uses().size() == 0) { - // LOG(INFO) << "destroy dictConstruct node now: " << node_info(producer_node); - // producer_node->destroy(); - // } - return true; - } - - bool eliminate_dict_getitems(Block* block) { - bool changed = false; - for (auto it = block->nodes().begin(); it != block->nodes().end();) { - // we might destroy the current node, so we need to pre-increment the iterator - Node* node = *it; - ++it; - for (Block* subblock : node->blocks()) { - changed |= eliminate_dict_getitems(subblock); - } - if (is_dict_getitem_node(node)) { - LOG(INFO) << "meet dict getitem after construct node :" << node_info(node); - changed |= eliminate_dict_getitem_after_construct(node); - } - } - return changed; - } - - std::shared_ptr graph_; -}; - -} // namespace - -void eliminate_some_dict(std::shared_ptr graph) { - EliminateSomeDict esd(std::move(graph)); - esd.run(); -} - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/lowering/eliminate_some_list.cpp b/poros/poros/lowering/eliminate_some_list.cpp deleted file mode 100644 index 72925bb3ec2..00000000000 --- a/poros/poros/lowering/eliminate_some_list.cpp +++ /dev/null @@ -1,340 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file eliminate_some_list.cpp -* @author tianjinjin@baidu.com -* @date Thu Sep 23 11:15:49 CST 2021 -* @brief -**/ -#include "poros/lowering/lowering_pass.h" - -#include -#include -#include -#include -#include - -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -namespace { -using namespace torch::jit; - -struct EliminateSomeList { - EliminateSomeList(std::shared_ptr graph) : graph_(std::move(graph)) {} - - void run() { - GRAPH_DUMP("before eliminate_some_lists Graph: ", graph_); - bool changed = eliminate_list_unpacks(graph_->block()); - if (changed) { - ConstantPropagation(graph_); - EliminateDeadCode(graph_); - EliminateCommonSubexpression(graph_); - ConstantPooling(graph_); - } - changed = eliminate_list_getitems(graph_->block()); - if (changed) { - ConstantPropagation(graph_); - EliminateDeadCode(graph_); - EliminateCommonSubexpression(graph_); - ConstantPooling(graph_); - } - GRAPH_DUMP("after eliminate_some_lists Graph: ", graph_); - return; - } - -private: - /* - * @brief - * 将以下graph: - * %reslist = prim::ListConstruct(%1, %2, %3) - * %res1, %res2, %res3 = prim::ListUnpack(%reslist) - * %4 = aten::matmul(%res1, %const.0) - * %5 = aten::matmul(%res2, %const.1) - * %6 = aten::matmul(%res3, %const.2) - * - * 或者以下graph: - * %reslist = prim::ListConstruct(%1) - * %reslist.2 = aten::append(%reslist, %2) - * %reslist.3 = aten::append(%reslist, %3) - * %res1, %res2, %res3 = prim::ListUnpack(%reslist) - * %4 = aten::matmul(%res1, %const.0) - * %5 = aten::matmul(%res2, %const.1) - * %6 = aten::matmul(%res3, %const.2) - * - * 替换成以下graph: - * %4 = aten::matmul(%1, %const.0) - * %5 = aten::matmul(%2, %const.1) - * %6 = aten::matmul(%3, %const.1) - * - * 需要特别注意的是: - * 1. 如果是ListConstruct + append 的模式,这些节点都得在同一个block,否则如果append在subblock下面,难以确定append的次数。 - * 2. ListConstruct 和 ListUnpack 间,除了同block下的append之外,不应该有其他可能改变该list的算子出现,否则会带来非预期的影响。 - */ - bool is_list_unpack_pattern(Node* node) { - if (node->kind() != prim::ListUnpack) { - return false; - } - - auto input_value = node->inputs().at(0); - auto producer_node = node->inputs().at(0)->node(); - - if (producer_node->kind() != prim::ListConstruct || - producer_node->owningBlock() != node->owningBlock()) { - return false; - } - - for (auto &use: input_value->uses()) { - if (use.user->kind() == aten::append) { - if (use.user->output()->uses().size() != 0 || - use.user->owningBlock() != node->owningBlock()) { - //aten::apend 的output有被其他节点用,或者不在一个block - LOG(INFO) << "find unmatched pattern: " << node_info(node); - return false; - } - } else if (use.user->kind() == prim::ListUnpack) { - continue; - } else { - //不满足我们找寻的条件 - LOG(INFO) << "find unmatched pattern: " << node_info(node); - return false; - } - } - return true; - } - - bool eliminate_list_unpack_after_construct(Node* node) { - //auto input_value = node->inputs().at(0); - auto producer_node = node->inputs().at(0)->node(); - - auto output_num = node->outputs().size(); - auto input_num = producer_node->inputs().size() + node->inputs()[0]->uses().size() - 1; - if (input_num != output_num) { - LOG(WARNING) << "ListConstruct + aten::append input_num not equal prim::ListUnpack output_num, " - << "bypass this node: " << node_info(node); - return false; - } - - std::vector input_value_list; - //prim::ListConstruct 的input 倒腾进去 - for (auto value : producer_node->inputs()) { - input_value_list.push_back(value); - } - - //TODO: 是否要排个序?? - std::vector append_node_list; - for (auto &use: node->inputs()[0]->uses()) { - if (use.user->kind() == aten::append) { - //aten::append 的 第二个 input 倒腾进去 - input_value_list.push_back(use.user->inputs()[1]); - append_node_list.push_back(use.user); - } - } - - if (input_value_list.size() != output_num) { - LOG(WARNING) << "ListConstruct + aten::append input_num not equal prim::ListUnpack output_num, " - << "bypass this node: " << node_info(node); - return false; - } - - int index = 0; - //开启output value 替换大法。 - for (auto output_value : node->outputs()) { - auto replace_value = input_value_list[index++]; - output_value->replaceAllUsesWith(replace_value); - } - - //本node可以destroy了。 - LOG(INFO) << "destroy listUnpack node now: " << node_info(node); - node->destroy(); - - //aten::append可以destroy了。 - for (auto &append_node: append_node_list) { - LOG(INFO) << "destroy aten::append node now: " << node_info(append_node); - append_node->destroy(); - } - - //producer_node可以destroy了。 - LOG(INFO) << "destroy listConstruct node now: " << node_info(producer_node); - producer_node->destroy(); - return true; - } - - /* - * @brief - * 将以下graph: - * %reslist = prim::ListConstruct(%1, %2, %3) - * %4 : int = prim::Constant[value=-1]() - * %res1 = aten::__getitem__(%reslist, %const.0) - * %5 = aten::matmul(%res1, %const.1) - * - * 或者以下graph: - * %reslist = prim::ListConstruct(%1) - * %reslist.2 = aten::append(%reslist, %2) - * %reslist.3 = aten::append(%reslist, %3) - * %4 : int = prim::Constant[value=-1]() - * %res1 = aten::__getitem__(%reslist, %4) - * %5 = aten::matmul(%res1, %const.1) - * - * 替换成以下graph: - * %5 = aten::matmul(%3, %const.1) - * - * 需要特别注意的是: - * 1. 如果有多个__getitem__的时候,append信息和listconstruct不能够删除, - * 当所有的__getitem__都处理完了才能删除append和listconstruct - * */ - bool is_list_getitem_node(Node* node) { - if (node->kind() != aten::__getitem__ || - node->inputs().at(1)->node()->kind() != prim::Constant) { - return false; - } - - auto producer_node = node->inputs().at(0)->node(); - if (producer_node->kind() != prim::ListConstruct || - producer_node->owningBlock() != node->owningBlock()) { - return false; - } - - return true; - } - - bool eliminate_list_getitem_after_construct(Node* node) { - auto input_value = node->inputs().at(0); - auto producer_node = input_value->node(); - - int get_item_count = 0; - for (auto &use: input_value->uses()) { - if (use.user->kind() == aten::append) { - if (use.user->output()->uses().size() != 0 || - use.user->owningBlock() != node->owningBlock()) { - //aten::apend 的output有被其他节点用,或者不在一个block - LOG(INFO) << "find unmatched pattern: " << node_info(node); - return false; - } - } else if (use.user->kind() == aten::__getitem__) { - get_item_count++; - continue; - } else { - //不满足我们找寻的条件 - LOG(INFO) << "find unmatched pattern: " << node_info(node); - return false; - } - } - - LOG(INFO) << "find list getitem after construct pattern: " << node_info(node); - auto input_num = producer_node->inputs().size() + node->inputs()[0]->uses().size() - 1; - - std::vector input_value_list; - //prim::ListConstruct 的input 倒腾进去 - for (auto value : producer_node->inputs()) { - input_value_list.push_back(value); - } - - //TODO: 是否要排个序?? - std::vector append_node_list; - for (auto &use: node->inputs()[0]->uses()) { - if (use.user->kind() == aten::append) { - //aten::append 的 第二个 input 倒腾进去 - input_value_list.push_back(use.user->inputs()[1]); - append_node_list.push_back(use.user); - } - } - - //求取index的值。 - int64_t index = toIValue((node->inputs()[1])->node()->output()).value().toInt(); - index = index < 0 ? input_num + index : index; - LOG(INFO) << "calculate getitem index number is : " << index; - - //开启output value 替换大法。 - node->outputs()[0]->replaceAllUsesWith(input_value_list[index]); - - //本node可以destroy了。 - LOG(INFO) << "destroy getitem node now: " << node_info(node); - node->destroy(); - - //当只有一个 getitem 节点的时候,aten::append 和 producer 都可以destroy了。 - if (get_item_count == 1) { - for (auto &append_node : append_node_list) { - LOG(INFO) << "destroy aten::append node now: " << node_info(append_node); - append_node->destroy(); - } - /* 以下的迭代方式可能出core。 - auto users_count = producer_node->output()->uses().size(); - for (int user_index = users_count; user_index >= 0; user_index--) { - auto append_node = producer_node->output()->uses()[user_index].user; - if (append_node->kind() == aten::append) { - LOG(INFO) << "destroy aten::append node now: " << node_info(append_node); - append_node->destroy(); - } - }*/ - //producer_node可以destroy了。 - LOG(INFO) << "destroy listConstruct node now: " << node_info(producer_node); - producer_node->destroy(); - } - return true; - } - - bool eliminate_list_unpacks(Block* block) { - bool changed = false; - for (auto it = block->nodes().begin(); it != block->nodes().end();) { - // we might destroy the current node, so we need to pre-increment - // the iterator - Node* node = *it; - ++it; - for (Block* subblock : node->blocks()) { - changed |= eliminate_list_unpacks(subblock); - } - if (is_list_unpack_pattern(node)) { - LOG(INFO) << "find list unpack after construct pattern: " << node_info(node); - changed |= eliminate_list_unpack_after_construct(node); - } - } - return changed; - } - - bool eliminate_list_getitems(Block* block) { - bool changed = false; - for (auto it = block->nodes().begin(); it != block->nodes().end();) { - // we might destroy the current node, so we need to pre-increment - // the iterator - Node* node = *it; - ++it; - for (Block* subblock : node->blocks()) { - changed |= eliminate_list_getitems(subblock); - } - if (is_list_getitem_node(node)) { - //LOG(INFO) << "meet list getitem after construct node :" << node_info(node); - changed |= eliminate_list_getitem_after_construct(node); - } - } - return changed; - } - -std::shared_ptr graph_; -}; - -} // namespace - -void eliminate_some_list(std::shared_ptr graph) { - EliminateSomeList esl(std::move(graph)); - esl.run(); -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/lowering/eliminate_subgraph_uesless_nodes.cpp b/poros/poros/lowering/eliminate_subgraph_uesless_nodes.cpp deleted file mode 100644 index 29ef7d904d9..00000000000 --- a/poros/poros/lowering/eliminate_subgraph_uesless_nodes.cpp +++ /dev/null @@ -1,151 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file eliminate_subgraph_useless_nodes.cpp -* @author tianshaoqing@baidu.com -* @date Thu May 16 19:49:02 CST 2022 -* @brief -**/ -#include "poros/lowering/lowering_pass.h" - -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -using namespace torch::jit; - -bool eliminate_subgraph_useless_nodes(std::shared_ptr subgraph, - torch::jit::Node& subgraph_node, - const bool is_input) { - AT_ASSERT(subgraph_node.kind() == torch::jit::prim::CudaFusionGroup); - // init useless schema set - std::unordered_set useless_schema_set; - useless_schema_set.emplace(torch::jit::parseSchema( - "aten::to.device(Tensor self, Device device, ScalarType dtype, bool non_blocking=False," - " bool copy=False, MemoryFormat? memory_format=None) -> Tensor").operator_name()); - useless_schema_set.emplace(torch::jit::parseSchema( - "aten::to.prim_Device(Tensor(a) self, Device? device, int? dtype=None," - " bool non_blocking=False, bool copy=False) -> (Tensor(b|a))").operator_name()); - useless_schema_set.emplace(torch::jit::parseSchema("aten::contiguous(Tensor(a) self, *, " - "MemoryFormat memory_format=contiguous_format) -> Tensor(a)").operator_name()); - useless_schema_set.emplace(torch::jit::parseSchema("aten::dropout(Tensor input, float p, " - "bool train) -> Tensor").operator_name()); - useless_schema_set.emplace(torch::jit::parseSchema("aten::detach(Tensor(a) self) -> Tensor(a)").operator_name()); - useless_schema_set.emplace(torch::jit::parseSchema("aten::unsqueeze(Tensor(a) self, int dim) -> Tensor(a)")\ - .operator_name()); - - // 由于execute_engine会对cpu的input做to(cuda)操作,所以子图输入前的to(cuda)可以删掉 - if (is_input) { - // 先处理子图输入是aten::to.device的 - at::ArrayRef node_inputs = subgraph_node.inputs(); - for (size_t i = 0; i < node_inputs.size(); i++) { - torch::jit::Node* maybe_to_device_node = node_inputs[i]->node(); - // 如果在set中找到了aten::to.device,其type参数不是默认,则不能删 - if (maybe_to_device_node->kind() == torch::jit::aten::to && - useless_schema_set.count(maybe_to_device_node->schema().operator_name()) != 0 && - maybe_to_device_node->input(2)->node()->kind() == torch::jit::prim::Constant && - toIValue(maybe_to_device_node->inputs()[2])->isNone()) { - - auto to_device_users = maybe_to_device_node->output(0)->uses(); - // 需要保证aten::to.device output的所有user只有prim::CudaFusionGroup这一种 - bool all_users_cfg = true; - for (size_t u = 0; u < to_device_users.size(); u++) { - if (to_device_users[u].user->kind() != prim::CudaFusionGroup) { - all_users_cfg = false; - break; - } - } - if (!all_users_cfg) { - continue; - } - // 给所有使用aten::to.device的子图替换输入 - for (size_t u = 0; u < to_device_users.size(); u++) { - to_device_users[u].user->replaceInput(to_device_users[u].offset, maybe_to_device_node->input(0)); - LOG(INFO) << "Remove aten::to.device input[" << i << "] of subgraph: " << - node_info(to_device_users[u].user) << ", which is useless."; - } - LOG(INFO) << "Destory node schema: [ " << maybe_to_device_node->schema() << " ]"; - // 删除aten::to.device - maybe_to_device_node->destroy(); - } - } - } else { - int unconst_nodes_num = 0; - // 删除子图内部的aten::to.device - auto cudafusion_subblock_nodes = subgraph->block()->nodes(); - for (auto c_it = cudafusion_subblock_nodes.begin(); c_it != cudafusion_subblock_nodes.end(); ) { - torch::jit::Node* maybe_useless_node = *c_it; - c_it++; - if (maybe_useless_node->kind() != torch::jit::prim::Constant) { - unconst_nodes_num++; - } - // 存在schema && 在useless_schema_set之中 - if (maybe_useless_node->maybeSchema() && - useless_schema_set.count(maybe_useless_node->schema().operator_name()) != 0) { - bool is_useless_node = false; - // 如果是aten::to.device,则需要额外判断scalartype是否为none,否则不能删 - if (maybe_useless_node->kind() == torch::jit::aten::to) { - if (maybe_useless_node->input(2)->node()->kind() == torch::jit::prim::Constant && - toIValue(maybe_useless_node->inputs()[2])->isNone()) { - is_useless_node = true; - } - // 对 rank=1 的 tensor 进行 aten::select 后接 aten::unsqueeze 的情况, - // 原本 torch 中 rank=1 的 tensor select后 rank 会等于 0, - // 有的模型(例如:faster-rcnn)会再加一次 unsqueeze 变回 rank=1,再进行其他操作。 - // 而 poros aten::select 的实现输出 nvtensor rank 依然是1,因此再 unsqueeze rank=2 就会出错。 - // 所以在子图里删掉这种情况的 aten::unsqueeze - } else if (maybe_useless_node->kind() == torch::jit::aten::unsqueeze && - maybe_useless_node->inputs().size() == 2 && - maybe_useless_node->input(1)->node()->kind() == torch::jit::prim::Constant) { - int64_t unsqueeze_dim = toIValue(maybe_useless_node->input(1)).value().toInt(); - torch::jit::Node* input0_node = maybe_useless_node->input(0)->node(); - if (input0_node->kind() == torch::jit::aten::select && - input0_node->outputs().size() == 1 && - input0_node->output(0)->type()->isSubtypeOf(c10::TensorType::get()) && - unsqueeze_dim == 0) { - auto select_output_type = input0_node->output(0)->type()->cast(); - // 通过c10::TensorType求rank - if (select_output_type->sizes().size().value() == 0) { - is_useless_node = true; - } - } - } else { - // 其他节点暂时不用判断 - is_useless_node = true; - } - - if (is_useless_node) { - LOG(INFO) << "Remove " << node_info(maybe_useless_node) << " in subgraph: "<< - node_info(&subgraph_node) << ", which is useless."; - LOG(INFO) << "Destory node schema: [ "<< maybe_useless_node->schema() << " ]"; - maybe_useless_node->output(0)->replaceAllUsesWith(maybe_useless_node->input(0)); - maybe_useless_node->destroy(); - unconst_nodes_num--; - } - } - } - // 如果删完子图中的无用节点后只有constant节点,则返回false unmerge。 - if (unconst_nodes_num <= 0) { - return false; - } - } - return true; -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/lowering/eliminate_useless_copy.cpp b/poros/poros/lowering/eliminate_useless_copy.cpp deleted file mode 100644 index adfd48e214b..00000000000 --- a/poros/poros/lowering/eliminate_useless_copy.cpp +++ /dev/null @@ -1,118 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file eliminate_useless_copy.cpp -* @author tianjinjin@baidu.com -* @date Thu Dec 16 16:27:02 CST 2021 -* @brief -**/ -#include "poros/lowering/lowering_pass.h" - -#include -#include -#include -#include -#include -#include - -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -namespace { -using namespace torch::jit; - -struct EliminateUselessCopy { - EliminateUselessCopy(std::shared_ptr graph) : graph_(std::move(graph)) {} - - void run() { - GRAPH_DUMP("before eliminate_useless_copys Graph: ", graph_); - bool changed = eliminate_useless_copys(graph_->block()); - if (changed) { - ConstantPropagation(graph_); - EliminateDeadCode(graph_); - EliminateCommonSubexpression(graph_); - ConstantPooling(graph_); - } - GRAPH_DUMP("after eliminate_useless_copys Graph: ", graph_); - return; - } - -private: - /* - * @brief - * 相关schema: aten::copy_(Tensor(a!) self, Tensor src, bool non_blocking=False) -> Tensor(a!) - * pytorch原始实现: https://github.com/pytorch/pytorch/blob/v1.9.0/aten/src/ATen/native/Copy.cpp#L246 - * - * 针对 %output = aten::copy_(%self, %src, %non_blocking) 形式的node。 - * 找出符合以下条件的aten::copy_ - * 1. %output 没有被其他node使用 - * 2. %self 除了aten::copy_ 本node外,没有被其他node使用 - * 当一个op同时满足以上两个条件时,认为该node可以直接删除。 - */ - bool is_node_useless_copy_pattern(Node* node) { - if (node->kind() != aten::copy_) { - return false; - } - - if (node->inputs().at(0)->uses().size() == 1 && - node->outputs().at(0)->uses().size() == 0) { - return true; - } - - LOG(INFO) << "find unmatched pattern: " << node_info(node); - return false; - } - - bool eliminate_useless_copy_node(Node* node) { - //本node可以destroy了。 - LOG(INFO) << "destroy aten::copy_ node now: " << node_info(node); - node->destroy(); - return true; - } - - bool eliminate_useless_copys(Block* block) { - bool changed = false; - for (auto it = block->nodes().rbegin(); it != block->nodes().rend();) { - // we might destroy the current node, so we need to pre-increment - // the iterator - Node* node = *it; - ++it; - for (Block* subblock : node->blocks()) { - changed |= eliminate_useless_copys(subblock); - } - if (is_node_useless_copy_pattern(node)) { - LOG(INFO) << "find useless aten copy pattern: " << node_info(node); - changed |= eliminate_useless_copy_node(node); - } - } - return changed; - } - -std::shared_ptr graph_; -}; - -} // namespace - -void eliminate_useless_copy(std::shared_ptr graph) { - EliminateUselessCopy euc(std::move(graph)); - euc.run(); -} - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/lowering/fuse_clip.cpp b/poros/poros/lowering/fuse_clip.cpp deleted file mode 100644 index dc1ddb6de4b..00000000000 --- a/poros/poros/lowering/fuse_clip.cpp +++ /dev/null @@ -1,88 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file fuse_clip.cpp -* @author tianshaoqing@baidu.com -* @date 2022-08-01 16:08:26 -* @brief -**/ - -#include "poros/lowering/fuse_clip.h" - -#include - -namespace baidu { -namespace mirana { -namespace poros { - -/** - * ReplaceClip - * @param graph - * @return true if graph changed, false if not - */ -bool FuseClip::fuse(std::shared_ptr graph) { - // schema: aten::clip(Tensor self, Scalar? min=None, Scalar? max=None) -> Tensor - if (try_to_replace_clip(graph->block())) { - std::string new_pattern = R"IR( - graph(%x, %min, %max): - %out : Tensor = aten::clamp(%x, %min, %max) - return (%out))IR"; - - std::string old_pattern = R"IR( - graph(%x, %min, %max): - %out : Tensor = aten::clip(%x, %min, %max) - return (%out))IR"; - - torch::jit::SubgraphRewriter std_rewriter; - std_rewriter.RegisterRewritePattern(old_pattern, new_pattern); - std_rewriter.runOnGraph(graph); - - return true; - } - return false; -} - -/** - * search for aten::clip recursively, record all findings - * @param block - * @return true if at least one aten::clip found, false if none found - */ -bool FuseClip::try_to_replace_clip(torch::jit::Block *block) { - bool graph_changed = false; - auto it = block->nodes().begin(); - while (it != block->nodes().end()) { - auto node = *it; - ++it; //++it first, node may be destroyed later。 - for (auto sub_block: node->blocks()) { - if (try_to_replace_clip(sub_block)) { - graph_changed = true; - } - } - //只处理 aten::clip场景 - if (node->kind() == torch::jit::aten::clip) { - graph_changed = true; - record_transform(torch::jit::aten::clip)->to(torch::jit::aten::clamp); - } - } - return graph_changed; -} - -FuseClip::FuseClip() = default; - -REGISTER_OP_FUSER(FuseClip) - -} -} -}// namespace \ No newline at end of file diff --git a/poros/poros/lowering/fuse_clip.h b/poros/poros/lowering/fuse_clip.h deleted file mode 100644 index e947ab0f4b0..00000000000 --- a/poros/poros/lowering/fuse_clip.h +++ /dev/null @@ -1,50 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file fuse_clip.h -* @author tianshaoqing@baidu.com -* @date 2022-08-01 16:08:26 -* @brief -**/ - -#pragma once - -#include "poros/lowering/op_fuse_pass.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class FuseClip : public IFuser { -public: - FuseClip(); - /** - * FuseClip - * @param graph - * @return true if graph changed, false if not - */ - bool fuse(std::shared_ptr graph); -private: - /** - * search for aten::clip recursively, record all findings - * @param block - * @return true if at least one clip found, false if none found - */ - bool try_to_replace_clip(torch::jit::Block *block); -}; - -} -} -} \ No newline at end of file diff --git a/poros/poros/lowering/fuse_conv_bn.cpp b/poros/poros/lowering/fuse_conv_bn.cpp deleted file mode 100644 index af195b20f4a..00000000000 --- a/poros/poros/lowering/fuse_conv_bn.cpp +++ /dev/null @@ -1,163 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file fuse_conv_bn.cpp -* @author Lin Xiao Chun (linxiaochun@baidu.com) -* @date 2022-03-31 16:11:19 -* @brief -**/ - -#include "poros/lowering/fuse_conv_bn.h" - -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -struct TORCH_API ConvBNParameters { - at::Tensor conv_w; - at::Tensor conv_b; - at::Tensor bn_rm; - at::Tensor bn_rv; - double bn_eps = 0.0; - at::Tensor bn_w; - at::Tensor bn_b; -}; - -// calculate weights and bias -std::tuple CalcFusedConvWeightAndBias( - const ConvBNParameters &p) { - at::Tensor bn_var_rsqrt = at::rsqrt(p.bn_rv + p.bn_eps); - const int64_t ndim = p.conv_w.dim(); - at::DimVector sizes(ndim, 1); - sizes.at(0) = -1; - at::Tensor new_w = p.conv_w * (p.bn_w * bn_var_rsqrt).reshape(sizes); - at::Tensor new_b = (p.conv_b - p.bn_rm) * bn_var_rsqrt * p.bn_w + p.bn_b; - return std::make_tuple(new_w, new_b); -} - -bool FuseConvBatchNorm::fuse(std::shared_ptr graph) { - graph_ = graph; - return try_to_fuse_conv_batchnorm(graph_->block()); -} - -FuseConvBatchNorm::FuseConvBatchNorm() = default; - -bool FuseConvBatchNorm::try_to_fuse_conv_batchnorm(torch::jit::Block *block) { - bool graph_changed = false; - auto it = block->nodes().begin(); - while (it != block->nodes().end()) { - auto node = *it; - ++it; //++it first, node may be destroyed later。 - for (auto sub_block: node->blocks()) { - if (try_to_fuse_conv_batchnorm(sub_block)) { - graph_changed = true; - } - } - //只处理 aten::conv + batch_norm的场景 - if (node->kind() != torch::jit::aten::batch_norm) { - continue; - } - - auto all_users = node->inputs()[0]->uses(); - if (all_users.size() != 1 || ((node->inputs()[0])->node()->kind() != torch::jit::aten::conv1d && - (node->inputs()[0])->node()->kind() != torch::jit::aten::conv2d && - (node->inputs()[0])->node()->kind() != torch::jit::aten::conv3d && - (node->inputs()[0])->node()->kind() != torch::jit::aten::_convolution)) { - continue; - } - - auto bn = node; - auto conv = (node->inputs()[0])->node(); - - // More parameters need to be checked when node is aten::_convolution. - if (conv->schema().operator_name() == torch::jit::parseSchema("aten::_convolution(Tensor input, Tensor weight, " - "Tensor? bias, int[] stride, int[] padding, int[] dilation, bool transposed, int[] output_padding, int groups, " - "bool benchmark, bool deterministic, bool cudnn_enabled, bool allow_tf32) -> Tensor ").operator_name()) { - bool transposed = toIValue(conv->input(6)).value().toBool(); - // deconvolution is not supported. - if (transposed) { - LOG(INFO) << "It is found that the transposed of aten::_convolution is true, which is not support to fuse conv+bn currently."; - continue; - } - // output_padding is not supported. - std::vector output_padding = toIValue(conv->input(7)).value().toIntVector(); - for (int64_t o : output_padding) { - if (o != 0) { - LOG(INFO) << "It is found that the output_padding of aten::_convolution is not equal to zero, " - "which is not support to fuse conv+bn currently."; - continue; - } - } - // other parameters like benchmark, deterministic, cudnn_enabled and allow_tf do not need to be checked for now. - } - - ConvBNParameters params; - // conv weights and bias - if (!(conv->inputs()[1])->type()->isSubtypeOf(c10::TensorType::get()) || //conv_weight - // !(conv->inputs()[2])->type()->isSubtypeOf(c10::TensorType::get()) || //conv_bias (maybe is None) - !(bn->inputs()[1])->type()->isSubtypeOf(c10::TensorType::get()) || //bn_weight - !(bn->inputs()[2])->type()->isSubtypeOf(c10::TensorType::get()) || //bn_bias - !(bn->inputs()[3])->type()->isSubtypeOf(c10::TensorType::get()) || //bn_mean - !(bn->inputs()[4])->type()->isSubtypeOf(c10::TensorType::get()) || //bn_var - !(bn->inputs()[7])->type()->isSubtypeOf(c10::FloatType::get())) { //bn_esp(default=1e-5) - continue; - } - - // record the fusing ops for debug - record_transform(conv, bn)->to(conv); - - params.conv_w = toIValue(conv->inputs()[1]).value().toTensor(); - if (toIValue(conv->inputs()[2]).value().isNone()) { - params.conv_b = torch::zeros({params.conv_w.size(0)}, {params.conv_w.device()}).to(params.conv_w.type()); - } else { - params.conv_b = toIValue(conv->inputs()[2]).value().toTensor(); - } - params.bn_w = toIValue(bn->inputs()[1]).value().toTensor(); - params.bn_b = toIValue(bn->inputs()[2]).value().toTensor(); - params.bn_rm = toIValue(bn->inputs()[3]).value().toTensor(); - params.bn_rv = toIValue(bn->inputs()[4]).value().toTensor(); - params.bn_eps = toIValue(bn->inputs()[7]).value().toDouble(); - - // calc new weights and bias - auto w_b = CalcFusedConvWeightAndBias(params); - - at::Tensor weights = std::get<0>(w_b); - at::Tensor bias = std::get<1>(w_b); - - torch::jit::WithInsertPoint guard(graph_->block()->nodes().front()); - auto conv_w = graph_->insertConstant(weights); - auto conv_b = graph_->insertConstant(bias); - conv_w->node()->moveBefore(conv); - conv_b->node()->moveBefore(conv); - - conv->replaceInput(1, conv_w); - conv->replaceInput(2, conv_b); - - bn->output()->replaceAllUsesWith(conv->output()); - bn->removeAllInputs(); - bn->destroy(); - - graph_changed = true; - } - return graph_changed; -} - -REGISTER_OP_FUSER(FuseConvBatchNorm) - -} -} -}// namespace \ No newline at end of file diff --git a/poros/poros/lowering/fuse_conv_bn.h b/poros/poros/lowering/fuse_conv_bn.h deleted file mode 100644 index 2c7e612127a..00000000000 --- a/poros/poros/lowering/fuse_conv_bn.h +++ /dev/null @@ -1,44 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file fuse_conv_bn.h -* @author Lin Xiao Chun (linxiaochun@baidu.com) -* @date 2022-03-31 16:11:19 -* @brief -**/ - -#pragma once - -#include "poros/lowering/op_fuse_pass.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class FuseConvBatchNorm : public IFuser { -public: - FuseConvBatchNorm(); - - bool fuse(std::shared_ptr graph); - -private: - bool try_to_fuse_conv_batchnorm(torch::jit::Block *block); - - std::shared_ptr graph_; -}; - -} -} -} \ No newline at end of file diff --git a/poros/poros/lowering/fuse_conv_mul.cpp b/poros/poros/lowering/fuse_conv_mul.cpp deleted file mode 100644 index 4e7863a8a4c..00000000000 --- a/poros/poros/lowering/fuse_conv_mul.cpp +++ /dev/null @@ -1,117 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file: fuse_conv_mul2.cpp -* @author: zhangfan51@baidu.com -* @data: 2022-04-24 18:43:02 -* @brief: -**/ -#include "poros/lowering/fuse_conv_mul.h" - -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -using namespace torch::jit; - -FuseConvMul::FuseConvMul() = default; - -/** - * FuseConvMul - * @param graph - * @return true if graph changed, false if not - */ -bool FuseConvMul::fuse(std::shared_ptr graph) { - graph_ = graph; - return try_to_fuse_conv_mul(graph_->block()); -} - -/** - * search for aten::conv + aten::mul patten for fuse - * @param block - * @return true if fuse success - */ -bool FuseConvMul::try_to_fuse_conv_mul(torch::jit::Block *block) { - bool graph_changed = false; - auto it = block->nodes().begin(); - while (it != block->nodes().end()) { - auto node = *it; - ++it; //先++it, node 可能destroy掉。 - for (auto sub_block : node->blocks()) { - if (try_to_fuse_conv_mul(sub_block)) { - graph_changed = true; - } - } - // find the op by "aten::mul".Scalar(Tensor self, Scalar other) -> Tensor" - if (node->kind() != aten::mul) { - continue; - } - - // find "aten::mul.Scalar(Tensor self, Scalar other) -> Tensor" - if (!node->inputs()[0]->type()->isSubtypeOf(c10::TensorType::get()) || - node->inputs()[1]->node()->kind() != prim::Constant) { - continue; - } - - auto conv = node->inputs()[0]->node(); - if((conv->kind() != aten::conv2d && conv->kind() != aten::conv1d && conv->kind() != aten::conv3d) || - node->inputs()[0]->uses().size() != 1) { - continue; - } - - if (!(conv->inputs()[1])->type()->isSubtypeOf(c10::TensorType::get()) || // conv_weight - conv->inputs()[1]->uses().size() != 1) { - continue; - } - at::Tensor conv_w = toIValue(conv->inputs()[1])->toTensor(); - float scale = toIValue(node->inputs()[1])->toScalar().toFloat(); - - torch::jit::WithInsertPoint guard(graph_->block()->nodes().front()); - // check bias - if (conv->inputs()[2]->type()->isSubtypeOf(c10::TensorType::get())) { - if (conv->inputs()[2]->uses().size() != 1) { - continue; - } - at::Tensor conv_b = toIValue(conv->inputs()[2])->toTensor(); - auto new_conv_b = graph_->insertConstant(conv_b * scale); - new_conv_b->setDebugName(conv->inputs()[2]->debugName() + ".scale"); - // 替换conv的bias值 - conv->inputs().at(2)->replaceAllUsesWith(new_conv_b); - } - - auto new_conv_w = graph_->insertConstant(conv_w * scale); - new_conv_w->setDebugName(conv->inputs()[1]->debugName() + ".scale"); - - // 替换conv的weight值 - conv->inputs().at(1)->replaceAllUsesWith(new_conv_w); - // 把所有的aten::mul的output的users更改为conv的output - node->output()->replaceAllUsesWith(conv->output()); - - LOG(INFO) << "Found fuse_conv2d_mul, node = " << *node; - // 删除 aten::mul节点 - node->removeAllInputs(); - node->destroy(); - graph_changed = true; - } - return graph_changed; -} - -REGISTER_OP_FUSER(FuseConvMul) - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/lowering/fuse_conv_mul.h b/poros/poros/lowering/fuse_conv_mul.h deleted file mode 100644 index ad67fc34dcd..00000000000 --- a/poros/poros/lowering/fuse_conv_mul.h +++ /dev/null @@ -1,53 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file: fuse_conv_mul.h -* @author: zhangfan51@baidu.com -* @data: 2022-04-24 18:41:20 -* @brief: -**/ - -#pragma once - -#include "poros/lowering/op_fuse_pass.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/** - * %3 : int = prim::Constant[value=1]() - * %4 : float = prim::Constant[value=2.0]() - * %1 : Tensor = aten::conv2d(%0, %conv_w, %conv_b, %conv_stride, %conv_padding, %conv_dilation, %3) - * %2 : Tensor = aten::mul(%1, %4) - * - * 如上面的IR,FuseConvMul是将conv + mul中的mul融到conv中,减少一次mul计算,融后在图上可以匹配到更多的针对conv的优化pass; - * 限制:aten::mul的输入%4需为constant类型。 - */ -class FuseConvMul : public IFuser { -public: - FuseConvMul(); - - bool fuse(std::shared_ptr graph); - -private: - bool try_to_fuse_conv_mul(torch::jit::Block *block); - - std::shared_ptr graph_; -}; - -} -} -} diff --git a/poros/poros/lowering/fuse_copy.cpp b/poros/poros/lowering/fuse_copy.cpp deleted file mode 100644 index de94bbc328a..00000000000 --- a/poros/poros/lowering/fuse_copy.cpp +++ /dev/null @@ -1,502 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file: fuse_copy.cpp -* @author: tianjinjin@baidu.com -* @data: Wed Jun 16 20:28:36 CST 2021 -* @brief: -**/ - -#include "poros/lowering/fuse_copy.h" - -#include -#include -#include -#include -#include -#include -#include -#include -#include - -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -using namespace torch::jit; - -FuseCopy::FuseCopy() = default; - -/** - * FuseCopy - * @param graph - * @return true if graph changed, false if not - */ -bool FuseCopy::fuse(std::shared_ptr graph) { - graph_ = graph; - GRAPH_DUMP("before fuse copy ops Graph: ", graph_); - bool fused = try_to_fuse_copy(graph_->block()); - if (fused) { - ConstantPropagation(graph_); - EliminateDeadCode(graph_); - EliminateCommonSubexpression(graph_); - ConstantPooling(graph_); - //EraseNumberTypesOnBlock(graph_->block); - //EliminateDeadCode(graph_->block, true, DCESideEffectPolicy::ALLOW_DELETING_NODES_WITH_SIDE_EFFECTS); - } - GRAPH_DUMP("after fuse copy ops Graph: ", graph_); - return fused; -} - -/** -* @brief 创建aten::size节点,用于生成给定dim的size信息 -* **/ -Value* FuseCopy::create_size_of_dim(Value* input, int64_t dim, Node* insertBefore) { - auto graph = input->owningGraph(); - WithInsertPoint guard(insertBefore); - auto size = graph->insert(aten::size, {input, dim}); - LOG(INFO) << "create_size_of_dim before node: " << node_info(insertBefore); - LOG(INFO) << "create aten::size node: " << node_info(size->node()); - return size; -} - -/** -* @brief 对value进行处理,尤其是需要对维度进行补充的情况(也就是被select给降维的情况需要把对应的维度补回来)。 -* **/ -void FuseCopy::adjust_value(Graph* graph, - Node* index_put_node, - const std::vector& slice_and_select_nodes, - Value* orig_data) { - //获取常量value的rank信息,如果value的rank为0或者为1,则不需要专门处理(虽然也可以在这里处理...) - //如果rank不是0,则这个tensor可能是select 生成的,需要提升维度,使得跟self维度一致后,再broadcast。 - bool need_unsqueeze_value = true; - Value* value = index_put_node->inputs().at(2); - if (value->node()->kind() == prim::Constant) { - at::Tensor value_tensor = toIValue(value).value().toTensor(); - int64_t value_rank = value_tensor.dim(); - if (value_rank == 0 || value_rank == 1) { - need_unsqueeze_value = false; - } - } - - if (need_unsqueeze_value == true) { - int64_t dim_offset = 0; - for (auto it = slice_and_select_nodes.rbegin(); it != slice_and_select_nodes.rend(); ++it) { - Node* node = *it; - int64_t dim = toIValue(node->inputs().at(1)).value().toInt(); - if (dim < 0) { - std::shared_ptr input_type = orig_data->type()->expect(); - if (input_type->dim().has_value()) { - int64_t rank = static_cast(input_type->dim().value()); - dim = dim + rank - dim_offset; - } - } - dim = dim + dim_offset; - - if (node->kind() == aten::select) { - //需要对value进行维度的还原。 - WithInsertPoint guard(index_put_node); - Value* unsqueeze = graph->insert(aten::unsqueeze, {index_put_node->inputs().at(2), dim}); - LOG(INFO) << "create aten::unsqueeze node: " << node_info(unsqueeze->node()); - index_put_node->replaceInput(2, unsqueeze); - dim_offset++; - } - } - } - return; -} - -/** -* @brief 创建aten::tensor节点包装indices信息。 -* **/ -Value* FuseCopy::convert_select_to_index(Value* index, Node* insertBefore) { - // Create index tensor based on index input of aten::select node. - auto graph = insertBefore->owningGraph(); - WithInsertPoint guard(insertBefore); - Node* indices = graph->create(aten::tensor, { - index, - graph->insertConstant(c10::ScalarType::Long), - //graph->insertConstant(torch::Device(torch::DeviceType::CUDA, 0)), - graph->insertConstant(torch::Device(at::kCPU)), - graph->insertConstant(false)}); - - indices->copyMetadata(insertBefore); - indices->insertBefore(insertBefore); - LOG(INFO) << "convert_select_to_index before node: " << node_info(insertBefore); - LOG(INFO) << "create aten::tensor node: " << node_info(indices); - return indices->output(); -} - -/** -* @brief 提取slice节点中的dim,start,end,step等信息,转化成slice tensor -* **/ -Value* FuseCopy::convert_slice_to_index(Node* slice, Value* size, Node* insertBefore) { - // Create index tensor based on aten::slice node. - auto graph = slice->owningGraph(); - WithInsertPoint guard(insertBefore); - TORCH_INTERNAL_ASSERT((slice->inputs()).size() == 5); - auto start = slice->inputs()[2]; - auto end = slice->inputs()[3]; - auto step = slice->inputs()[4]; - //auto index = graph->insert(aten::arange, {size}); - auto index = graph->insert(aten::arange, {size}, {NamedValue("dtype", c10::kLong)}); - LOG(INFO) << "convert_slice_to_index before node: " << node_info(insertBefore); - LOG(INFO) << "create aten::arange node: " << node_info(index->node()); - auto sliced_index_n = graph->create(aten::slice, { - index, - graph->insertConstant(at::Scalar(0)), - start, - end, - step}); - LOG(INFO) << "create aten::slice node: " << node_info(sliced_index_n); - sliced_index_n->copyMetadata(insertBefore); - auto sliced_index = sliced_index_n->insertBefore(insertBefore)->output(); - return sliced_index; -} - -//torch.version >= 1.12, Source api发生调整,兼容之 -#if TORCH_VERSION_MAJOR >= 1 && TORCH_VERSION_MINOR >= 12 -#define NODE_SOURCE_TEXT(name) \ - name->text_str() -#else -#define NODE_SOURCE_TEXT(name) \ - name->text() -#endif - -/** -* @brief 找到跟 copy_ 或者 index_put_ 等op相关联的 slice op -* 他们来自python的同一行代码, -* 是为了合作完成list 或者 tensor 的切片功能 -* 比如 y = x[1:3, 0] 这样的形式 -// Example graph: -// %306 : Float(*, 16, 64, 16, 16) = aten::slice(%out.4, %0, %none, %none, %1) -// %307 : Float(*, 15, 64, 16, 16) = aten::slice(%306, %1, %none, %11, %1) -// %308 : Float(*, 15, 8, 16, 16) = aten::slice(%307, %2, %none, %y, %1) -// %309 : Tensor = aten::copy_(%308, %305, %false) -* **/ -std::vector FuseCopy::fetch_slice_and_select_pattern(const Node* node) { - TORCH_INTERNAL_ASSERT(node->kind() == aten::index_put || - node->kind() == aten::index_put_ || - node->kind() == aten::copy_); - const auto& node_source = node->sourceRange().source(); - - std::vector slice_and_select_nodes; - auto src_node = node->input(0)->node(); - while (src_node) { - auto& src_node_source = src_node->sourceRange().source(); - if ((src_node->kind() == aten::slice || src_node->kind() == aten::select) && - NODE_SOURCE_TEXT(node_source) == NODE_SOURCE_TEXT(src_node_source) && - node_source->starting_line_no() == src_node_source->starting_line_no()) { - slice_and_select_nodes.emplace_back(src_node); - //常常是连续的slice - src_node = src_node->input(0)->node(); - } else { - src_node = nullptr; - } - } - return slice_and_select_nodes; -} - -/** -* @brief 把相关联的slice 和 select 整合成 indices: -* **/ -std::unordered_map FuseCopy::merge_slice_and_select_to_indices( - Graph* graph, - Node* index_put_node, - const std::vector& slice_and_select_nodes, - Value* orig_data) { - - std::unordered_map dim_index_map; - int64_t cur_dim = 0; - /* dim_offset 的意义: 当select 和 slice 混合出现,完成对 tensor 的切片功能时, - 由于select 有降维的效果,aten::select 后面跟的 op (包括 select 和 slice) 的 dim 信息会被影响到, - 所以需要根据aten::select 已经出现的次数,对后续 op 的dim信息进行修正。 - */ - int64_t dim_offset = 0; - const auto orig_tensor_indices = index_put_node->input(1)->node()->inputs(); - // slice_and_select_nodes 的添加过程是逆序的,所以逆向迭代vector 内的 slice 和 select 节点。 - for (auto it = slice_and_select_nodes.rbegin(); it != slice_and_select_nodes.rend(); ++it) { - Node* node = *it; - LOG(INFO) << "handle slice or select node info: " << node_info(node); - //int64_t dim = node->inputs().at(1)->node()->t(attr::value).item().toLong(); - int64_t dim = toIValue(node->inputs().at(1)).value().toInt(); - if (dim < 0) { - auto input_type = orig_data->type()->expect(); - if (input_type->dim().has_value()) { - auto rank = static_cast(input_type->dim().value()); - dim = dim + rank - dim_offset; - } else { - std::cerr << "Error: Poros handle index Ops - Cannot export ellipsis indexing for input " - << "of unknown rank."; - } - } - - dim = dim + dim_offset; - while (cur_dim < dim) { - if (cur_dim - dim_offset >= (int64_t)orig_tensor_indices.size() || - index_put_node->input(1)->node()->input(cur_dim - dim_offset)->node()->mustBeNone()) { - auto size = create_size_of_dim(orig_data, cur_dim, index_put_node); - WithInsertPoint guard(index_put_node); - //auto index_tensor = graph->insert(aten::arange, {size}); - auto index_tensor = graph->insert(aten::arange, {size}, {NamedValue("dtype", c10::kLong)}); - LOG(INFO) << "create aten::arange node: " << node_info(index_tensor->node()); - dim_index_map.emplace(std::piecewise_construct, std::forward_as_tuple(cur_dim), - std::forward_as_tuple(index_tensor, aten::slice)); - } else if (cur_dim - dim_offset < (int64_t)orig_tensor_indices.size()) { - dim_index_map.emplace(std::piecewise_construct, std::forward_as_tuple(cur_dim), - std::forward_as_tuple(orig_tensor_indices[cur_dim - dim_offset], aten::index)); - } - cur_dim++; - } - - AT_ASSERT(cur_dim == dim); - LOG(INFO) << "cur_dim info: " << cur_dim << ", dim_offset: " << dim_offset; - - if (node->kind() == aten::slice) { - auto size = create_size_of_dim(orig_data, dim, index_put_node); - auto index_tensor = convert_slice_to_index(node, size, index_put_node); - dim_index_map.emplace(std::piecewise_construct, std::forward_as_tuple(dim), - std::forward_as_tuple(index_tensor, aten::slice)); - } else if (node->kind() == aten::select) { - auto index_tensor = convert_select_to_index(node->input(2), index_put_node); - dim_index_map.emplace(std::piecewise_construct, std::forward_as_tuple(dim), - std::forward_as_tuple(index_tensor, aten::select)); - dim_offset++; - } else { - AT_ERROR("Unexpected node kind ", node->kind().toDisplayString(), " Expected aten::slice or aten::select."); - } - cur_dim++; - } - - while (cur_dim - dim_offset < (int64_t)orig_tensor_indices.size()) { - dim_index_map.emplace(std::piecewise_construct, std::forward_as_tuple(cur_dim), - std::forward_as_tuple(orig_tensor_indices[cur_dim - dim_offset], aten::index)); - cur_dim++; - } - // Each dimension should have its associated index tensor. - AT_ASSERT((int64_t)dim_index_map.size() == cur_dim); - return dim_index_map; -} - -std::vector FuseCopy::reshape_to_advanced_indexing_format(Graph* graph, Node* index_put_node, - std::unordered_map& dim_index_map) { - std::vector indices; - size_t min_index_dim = dim_index_map.size(); - size_t max_index_dim = 0; - size_t tensor_ind_count = 0; - for (size_t i = 0; i < dim_index_map.size(); ++i) { - auto index_i = dim_index_map.find(i); - AT_ASSERT(index_i != dim_index_map.end()); - if (index_i->second.orig_node_kind == aten::index) { - if (i < min_index_dim) - min_index_dim = i; - if (i > max_index_dim) - max_index_dim = i; - tensor_ind_count++; - } - } - - if (((max_index_dim - min_index_dim + 1) != tensor_ind_count) && tensor_ind_count != 0) { - AT_ERROR("Only consecutive 1-d tensor indices are supported in exporting aten::index_put to POROS."); - } - - size_t tensor_ind_offset = tensor_ind_count == 0 ? 0 : tensor_ind_count - 1; - WithInsertPoint guard(index_put_node); - for (size_t i = 0; i < dim_index_map.size(); ++i) { - size_t ind_size = 0; - auto index_i = dim_index_map.find(i); - AT_ASSERT(index_i != dim_index_map.end()); - Value* index = index_i->second.index; - switch (index_i->second.orig_node_kind) { - case aten::select: - case aten::slice: { - if (i < min_index_dim) { - ind_size = dim_index_map.size() - tensor_ind_offset - i; - } else { - ind_size = dim_index_map.size() - i; - } - break; - } - case aten::index: { - ind_size = dim_index_map.size() - tensor_ind_offset - min_index_dim; - break; - } - default: - AT_ERROR("Unexpected node kind ", index_i->second.orig_node_kind); - } - - if (ind_size != 1) { - std::vector view_shape(ind_size, 1); - view_shape[0] = -1; - auto unsqueezed_index = graph->insert(aten::view, {index, view_shape}); - LOG(INFO) << "create aten::view node: " << node_info(unsqueezed_index->node()); - indices.emplace_back(unsqueezed_index); - } else { - indices.emplace_back(index); - } - } - return indices; -} - -/** -* @brief 针对aten::index_put / aten::index_put_ 的处理: -* 将跟他们相关联的slice 和 select 节点整合到一起, 提取indices信息,重新写个index_put。 -* **/ -bool FuseCopy::prepare_index_put(Node* index_put_node) { - LOG(INFO) << "prepare for index put node: " << node_info(index_put_node); - TORCH_INTERNAL_ASSERT(index_put_node->kind() == aten::index_put || - index_put_node->kind() == aten::index_put_); - //找到相关联的slice 和 select - std::vector slice_and_select_nodes = fetch_slice_and_select_pattern(index_put_node); - if (slice_and_select_nodes.size() == 0) { - return false; - } - LOG(INFO) << "slice_and_select_nodes_size: " << slice_and_select_nodes.size(); - Node* last_node = slice_and_select_nodes.size() > 0 ? slice_and_select_nodes.back() : index_put_node; - //找到最原始的那个被切片的value, 具体到example graph中,原始value 为 %out.4。 - Value* orig_data = last_node->input(0); - //当index_put 所在的node 与被改变的value不在一个block的时候,跳过这种情况。 - if (orig_data->node()->owningBlock() != index_put_node->owningBlock()) { - LOG(INFO) << "orig data comes from different block, bypass this situation"; - return false; - } - - auto graph = index_put_node->owningGraph(); - //对value进行处理。 - adjust_value(graph, index_put_node, slice_and_select_nodes, orig_data); - - //把slice和select操作转变成indices。 - std::unordered_map dim_index_map = - merge_slice_and_select_to_indices(graph, index_put_node, slice_and_select_nodes, orig_data); - - - std::vector indices = reshape_to_advanced_indexing_format(graph, index_put_node, dim_index_map); - - // Create new index_put node with converted indices. - const auto list_indices = graph->createList(OptionalType::ofTensor(), indices) - ->insertBefore(index_put_node)->output(); - LOG(INFO) << "create tensorlist node: " << node_info(list_indices->node()); - auto new_index_put_node = graph->create(aten::index_put, - {orig_data, list_indices, index_put_node->input(2), index_put_node->input(3)}); - LOG(INFO) << "create aten::index_put node: " << node_info(new_index_put_node); - new_index_put_node->insertBefore(index_put_node); - new_index_put_node->copyMetadata(index_put_node); - auto new_index_put = new_index_put_node->output(); - new_index_put->copyMetadata(index_put_node->output()); - index_put_node->output()->replaceAllUsesWith(new_index_put); - orig_data->replaceAllUsesAfterNodeWith(index_put_node, new_index_put); - record_transform(index_put_node)->to(new_index_put_node); - index_put_node->destroy(); - return true; -} - - -/** -* @brief 针对aten::copy_的处理: 将其用 index_put_ 代替。 -* 此步骤中用到的dummylist 只是一个”站位符“,不能真正用于index_put -* prepare_index_put 会找到真正的 index信息。 -* **/ - -// Example: -// %out: Tensor = aten::copy_(%self, %src, %non_blocking) -// -// After this prepare function: -// %dummylist : Tensor?[] = prim::ListConstruct() -// %newout: Tensor = aten::index_put_(%self, %dummylist, %src, %non_blocking) - bool FuseCopy::prepare_copy(Node* node) { - TORCH_INTERNAL_ASSERT(node->kind() == aten::copy_); - LOG(INFO) << "prepare for copy node: " << node_info(node); - - //找到相关联的slice 和 select - std::vector slice_and_select_nodes = fetch_slice_and_select_pattern(node); - if (slice_and_select_nodes.size() == 0) { - return false; - } - - //找到最原始的那个被切片的value, 具体到example graph中,原始value 为 %out.4。先解决引用语义的问题。 - Node* last_node = slice_and_select_nodes.back(); - Value* orig_data = last_node->input(0); - //当copy_ 所在的node 与被改变的value不在一个block的时候,跳过这种情况。 - if (orig_data->node()->owningBlock() != node->owningBlock()) { - LOG(INFO) << "orig data comes from different block, bypass this situation"; - return false; - } - orig_data->replaceAllUsesAfterNodeWith(node, node->output()); - - //做index_put 的替换 - WithInsertPoint guard(node); - auto graph = node->owningGraph(); - Value* dummy_list = graph->insertNode(graph->createList(OptionalType::ofTensor(), {}))->output(); - - // 当value的size跟self的size不一致的时候,需要对齐两者的size信息, - // 尝试在此处直接用expand_as, 发现单测无法通过,因为index_put支持value的rank为0的情况, - // 此时需要修改index_put converter 的实现,兼容value 的rank为0的情况。 - // Value* expanded_value = graph->insert(aten::expand_as, - // {node->input(1), orig_data}); - // expanded_value->node()->setSourceRange(node->sourceRange()); - // expanded_value->copyMetadata(node->input(1)); - // expanded_value->node()->copyMetadata(node); - - Value* index_put = graph->insert(aten::index_put_, - {node->input(0), dummy_list, node->input(1), node->input(2)}); - index_put->node()->copyMetadata(node); - index_put->copyMetadata(node->output()); - node->output()->replaceAllUsesWith(index_put); - - record_transform(node)->to(index_put->node()); - bool changed = prepare_index_put(index_put->node()); - if (changed == true) { - node->destroy(); - } - return changed; -} - -/** - * search for aten::copy_ or aten::index_put patten for fuse - * @param block - * @return true if fuse success - */ -bool FuseCopy::try_to_fuse_copy(torch::jit::Block *block) { - bool graph_changed = false; - for (auto it = block->nodes().begin(); it != block->nodes().end();) { - Node* node = *it; - it++; // node n can be destroyed - - auto nkind = node->kind(); - //sub_block situation - if (nkind == prim::If || nkind == prim::Loop) { - for (Block* sub_block : node->blocks()) { - try_to_fuse_copy(sub_block); - } - } else { - if (nkind == aten::copy_) { - LOG(INFO) << "copy situation meet"; - graph_changed |= prepare_copy(node); - } else if (nkind == aten::index_put || nkind == aten::index_put_) { - LOG(INFO) << "index_put or index_put situation meet"; - graph_changed |= prepare_index_put(node); - } - } - } - return graph_changed; -} - -REGISTER_OP_FUSER(FuseCopy) - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/lowering/fuse_copy.h b/poros/poros/lowering/fuse_copy.h deleted file mode 100644 index ee4617196a8..00000000000 --- a/poros/poros/lowering/fuse_copy.h +++ /dev/null @@ -1,136 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file: fuse_copy.h -* @author: tianjinjin@baidu.com -* @data: Mon Aug 22 11:33:45 CST 2022 -* @brief: -**/ - -#pragma once - -#include "poros/lowering/op_fuse_pass.h" - -namespace baidu { -namespace mirana { -namespace poros { - -/*** - * torchscript 中有个重要的概念,是view, - * 当slice这个op出现时(包括连续出现时),不会真正的执行内存的copy,而是通过view去尽可能的复用buffer - * 直到出现copy_ 才会进行真正的buffer的拷贝。 - * 比如下面的graph,执行到最后,真正发生了变化的是 %out.4。 - * typical graph for copy_ - - %none : NoneType = prim::Constant() - %false : bool = prim::Constant[value=0]() - %0 : int = prim::Constant[value=0]() - %1 : int = prim::Constant[value=1]() - %2 : int = prim::Constant[value=2]() - %11 : int = prim::Constant[value=-1]() - %out.4 : Float(*, 16, 64, 16, 16) = aten::zeros_like(%x, %none, %none, %none, %none, %none) - %303 : Float(*, 16, 64, 16, 16) = aten::slice(%x, %0, %none, %none, %1) - %304 : Float(*, 15, 64, 16, 16) = aten::slice(%303, %1, %1, %none, %1) - %305 : Float(*, 15, 8, 16, 16) = aten::slice(%304, %2, %none, %y, %1) - - %306 : Float(*, 16, 64, 16, 16) = aten::slice(%out.4, %0, %none, %none, %1) - %307 : Float(*, 15, 64, 16, 16) = aten::slice(%306, %1, %none, %11, %1) - %308 : Float(*, 15, 8, 16, 16) = aten::slice(%307, %2, %none, %y, %1) - %309 : Tensor = aten::copy_(%308, %305, %false) - - %310 : Float(*, 15, 64, 16, 16) = aten::slice(%303, %1, %none, %11, %1) - %311 : int = aten::mul(%2, %y) - %312 : Float(*, 15, 8, 16, 16) = aten::slice(%310, %2, %y, %311, %1) - %313 : Float(*, 16, 64, 16, 16) = aten::slice(%out.4, %0, %none, %none, %1) - %314 : Float(*, 15, 64, 16, 16) = aten::slice(%313, %1, %1, %none, %1) - %315 : Float(*, 15, 8, 16, 16) = aten::slice(%314, %2, %y, %311, %1) - %316 : Tensor = aten::copy_(%315, %312, %false) - - %317 : Float(*, 16, 64, 16, 16) = aten::slice(%303, %1, %none, %none, %1) - %318 : Float(*, 16, 48, 16, 16) = aten::slice(%317, %2, %311, %none, %1) - %319 : Float(*, 16, 64, 16, 16) = aten::slice(%out.4, %0, %none, %none, %1) - %320 : Float(*, 16, 64, 16, 16) = aten::slice(%319, %1, %none, %none, %1) - %321 : Float(*, 16, 48, 16, 16) = aten::slice(%320, %2, %311, %none, %1) - %322 : Tensor = aten::copy_(%321, %318, %false) - - %323 : int[] = prim::ListConstruct(%nt.3, %c.3, %h.3, %w.3) - %final : Float(*, 64, 16, 16) = aten::view(%out.4, %323) - * - * the implementation of index_put: - * aten/src/ATen/native/cuda/indexing.cu - * https://github.com/pytorch/pytorch/blob/v1.9.0-rc1/aten/src/ATen/native/cuda/Indexing.cu#L209 - * ***/ - -struct ConvertedIndex { - ConvertedIndex(torch::jit::Value* index, c10::Symbol orig_node_kind) - : index(index), orig_node_kind(orig_node_kind) {} - - torch::jit::Value* index = nullptr; - c10::Symbol orig_node_kind; -}; - -/** - * 目前可以处理的场景包括: - * 1. 纯slice的场景: out[:, :-1, :3] = x[:, 1:, :3] - * 2. slice + 单个select的场景(等号右侧为单值): out[2:3:1, :, :, 0, :] = 1 - * 3. slice + 多个select的场景(等号右侧为单值): out[2:3:1, 3, :, 0, :] = 1 - * 4. slice + 单个select的场景(等号右侧为tensor): boxes[:, :, 0] = torch.clamp(boxes[:, :, 0], min=0) - * 5. sclie + 多个select的场景(等号右侧为tensor): boxes[:, 0, :, 1] = torch.clamp(boxes[:, 0, :, 1], min=0) - * **/ -class FuseCopy : public IFuser { -public: - FuseCopy(); - - bool fuse(std::shared_ptr graph); - -private: - bool try_to_fuse_copy(torch::jit::Block *block); - - bool prepare_copy(torch::jit::Node* node); - bool prepare_index_put(torch::jit::Node* index_put_node); - - torch::jit::Value* create_size_of_dim(torch::jit::Value* input, - int64_t dim, - torch::jit::Node* insertBefore); - torch::jit::Value* convert_select_to_index(torch::jit::Value* index, - torch::jit::Node* insertBefore); - torch::jit::Value* convert_slice_to_index(torch::jit::Node* slice, - torch::jit::Value* size, - torch::jit::Node* insertBefore); - - std::vector fetch_slice_and_select_pattern(const torch::jit::Node* node); - - std::unordered_map merge_slice_and_select_to_indices( - torch::jit::Graph* graph, - torch::jit::Node* index_put_node, - const std::vector& slice_and_select_nodes, - torch::jit::Value* orig_data); - - std::vector reshape_to_advanced_indexing_format( - torch::jit::Graph* graph, - torch::jit::Node* index_put_node, - std::unordered_map& dim_index_map); - - void adjust_value(torch::jit::Graph* graph, - torch::jit::Node* index_put_node, - const std::vector& slice_and_select_nodes, - torch::jit::Value* orig_data); - - std::shared_ptr graph_; -}; - -} -} -} diff --git a/poros/poros/lowering/fuse_gelu.cpp b/poros/poros/lowering/fuse_gelu.cpp deleted file mode 100644 index 286bdec85f9..00000000000 --- a/poros/poros/lowering/fuse_gelu.cpp +++ /dev/null @@ -1,128 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file fuse_gelu.cpp -* @author tianshaoqing@baidu.com -* @date 2022-10-20 14:39:32 -* @brief -**/ - -#include "poros/lowering/fuse_gelu.h" - -#include -#include - -namespace baidu { -namespace mirana { -namespace poros { -/** - * Rewrite aten::gelu to the fast version: - * y = 0.5 * x * (1 + tanh(sqrt(2 / Pi) * (x + 0.044715 * x^3))) - * Note: This may result in a small diff. - * @param graph - * @return - */ -bool FuseGelu::fuse(std::shared_ptr graph) { - if (try_to_find_gelu(graph->block())) { - std::string gelu_pattern; - std::string gelu_reduce_pattern; - - if (TORCH_VERSION_MAJOR < 2 && TORCH_VERSION_MINOR < 12) { - gelu_pattern = R"IR( - graph(%x): - %out : Tensor = aten::gelu(%x) - return (%out))IR"; - - gelu_reduce_pattern = R"IR( - graph(%x.1 : Tensor): - %1 : float = prim::Constant[value=0.044714999999999998]() - %2 : float = prim::Constant[value=0.79788456080000003]() - %3 : int = prim::Constant[value=3]() - %4 : float = prim::Constant[value=1.0]() - %5 : float = prim::Constant[value=0.5]() - %6 : Tensor = aten::pow(%x.1, %3) - %7 : Tensor = aten::mul(%6, %1) - %8 : Tensor = aten::add(%7, %x.1, %4) - %9 : Tensor = aten::mul(%8, %2) - %10 : Tensor = aten::tanh(%9) - %11 : Tensor = aten::add(%10, %4, %4) - %12 : Tensor = aten::mul(%11, %x.1) - %13 : Tensor = aten::mul(%12, %5) - return (%13))IR"; - } else { - gelu_pattern = R"IR( - graph(%x : Tensor, %approximate : str): - %out : Tensor = aten::gelu(%x, %approximate) - return (%out))IR"; - - gelu_reduce_pattern = R"IR( - graph(%x.1 : Tensor, %approximate): - %1 : float = prim::Constant[value=0.044714999999999998]() - %2 : float = prim::Constant[value=0.79788456080000003]() - %3 : int = prim::Constant[value=3]() - %4 : float = prim::Constant[value=1.0]() - %5 : float = prim::Constant[value=0.5]() - %6 : Tensor = aten::pow(%x.1, %3) - %7 : Tensor = aten::mul(%6, %1) - %8 : Tensor = aten::add(%7, %x.1, %4) - %9 : Tensor = aten::mul(%8, %2) - %10 : Tensor = aten::tanh(%9) - %11 : Tensor = aten::add(%10, %4, %4) - %12 : Tensor = aten::mul(%11, %x.1) - %13 : Tensor = aten::mul(%12, %5) - return (%13))IR"; - } - torch::jit::SubgraphRewriter gelu_rewriter; - gelu_rewriter.RegisterRewritePattern(gelu_pattern, gelu_reduce_pattern); - gelu_rewriter.runOnGraph(graph); - - return true; - } - return false; -} - -FuseGelu::FuseGelu() = default; - -/** - * find out whether gelu exists. - * @param block - * @return bool: true if aten::gelu exists, else false. - */ -bool FuseGelu::try_to_find_gelu(torch::jit::Block *block) { - bool graph_changed = false; - auto it = block->nodes().begin(); - while (it != block->nodes().end()) { - auto node = *it; - ++it; //++it first, node may be destroyed later。 - for (auto sub_block: node->blocks()) { - if (try_to_find_gelu(sub_block)) { - graph_changed = true; - } - } - - if (node->kind() == torch::jit::aten::gelu) { - record_transform(torch::jit::aten::gelu)->to(torch::jit::aten::pow, torch::jit::aten::mul, - torch::jit::aten::add, torch::jit::aten::tanh); - graph_changed = true; - } - } - return graph_changed; -} - -REGISTER_OP_FUSER(FuseGelu) - -} -} -}// namespace \ No newline at end of file diff --git a/poros/poros/lowering/fuse_gelu.h b/poros/poros/lowering/fuse_gelu.h deleted file mode 100644 index f0fe5bf047a..00000000000 --- a/poros/poros/lowering/fuse_gelu.h +++ /dev/null @@ -1,42 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file fuse_gelu.h -* @author tianshaoqing@baidu.com -* @date 2022-10-20 14:39:32 -* @brief -**/ - -#pragma once - -#include "poros/lowering/op_fuse_pass.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class FuseGelu : public IFuser { -public: - FuseGelu(); - - bool fuse(std::shared_ptr graph); - -private: - bool try_to_find_gelu(torch::jit::Block *block); -}; - -} -} -} \ No newline at end of file diff --git a/poros/poros/lowering/fuse_hard_swish.cpp b/poros/poros/lowering/fuse_hard_swish.cpp deleted file mode 100644 index e4832a759f3..00000000000 --- a/poros/poros/lowering/fuse_hard_swish.cpp +++ /dev/null @@ -1,96 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file fuse_hard_swish.cpp -* @author Lin Xiao Chun (linxiaochun@baidu.com) -* @date 2022-04-07 15:30:35 -* @brief -**/ -#include "poros/lowering/fuse_hard_swish.h" - -#include - -namespace baidu { -namespace mirana { -namespace poros { - -/** - * FuseHardSwish - * @param graph - * @return true if graph changed, false if not - */ -bool FuseHardSwish::fuse(std::shared_ptr graph) { - if (try_to_find_hardswish(graph->block())) { - std::string new_pattern = R"IR( - graph(%x): - %1 : int = prim::Constant[value=1]() - %3 : int = prim::Constant[value=3]() - %6 : int = prim::Constant[value=6]() - %7 : int = prim::Constant[value=0]() - %x_1 : Tensor = aten::add(%x, %3, %1) - %x_2 : Tensor = aten::clamp(%x_1, %7, %6) - %x_3 : Tensor = aten::mul(%x, %x_2) - %out : Tensor = aten::div(%x_3, %6) - return (%out))IR"; - - std::string old_pattern = R"IR( - graph(%x): - %out: Tensor = aten::hardswish(%x) - return (%out))IR"; - - torch::jit::SubgraphRewriter std_rewriter; - std_rewriter.RegisterRewritePattern(old_pattern, new_pattern); - std_rewriter.runOnGraph(graph); - - return true; - } - - return false; -} - -/** - * search for hardswish activation recursively, record all findings - * @param block - * @return true if at least one hardswish found, false if none found - */ -bool FuseHardSwish::try_to_find_hardswish(torch::jit::Block *block) { - bool graph_changed = false; - auto it = block->nodes().begin(); - while (it != block->nodes().end()) { - auto node = *it; - ++it; //++it first, node may be destroyed later。 - for (auto sub_block: node->blocks()) { - if (try_to_find_hardswish(sub_block)) { - graph_changed = true; - } - } - //只处理 aten::hardswish的场景 - if (node->kind() != torch::jit::aten::hardswish) { - continue; - } - record_transform(torch::jit::aten::hardswish)->to(torch::jit::aten::add, torch::jit::aten::clamp, torch::jit::aten::div); - - graph_changed = true; - } - return graph_changed; -} - -FuseHardSwish::FuseHardSwish() = default; - -REGISTER_OP_FUSER(FuseHardSwish) - -} -} -}// namespace \ No newline at end of file diff --git a/poros/poros/lowering/fuse_hard_swish.h b/poros/poros/lowering/fuse_hard_swish.h deleted file mode 100644 index 3054b02bdf7..00000000000 --- a/poros/poros/lowering/fuse_hard_swish.h +++ /dev/null @@ -1,51 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file fuse_hard_swish.h -* @author Lin Xiao Chun (linxiaochun@baidu.com) -* @date 2022-04-07 15:31:26 -* @brief -**/ - -#pragma once - -#include "poros/lowering/op_fuse_pass.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class FuseHardSwish : public IFuser { -public: - FuseHardSwish(); - - /** - * FuseHardSwish - * @param graph - * @return true if graph changed, false if not - */ - bool fuse(std::shared_ptr graph); -private: - /** - * search for hardswish activation recursively, record all findings - * @param block - * @return true if at least one hardswish found, false if none found - */ - bool try_to_find_hardswish(torch::jit::Block *block); -}; - -} -} -} \ No newline at end of file diff --git a/poros/poros/lowering/fuse_meshgrid.cpp b/poros/poros/lowering/fuse_meshgrid.cpp deleted file mode 100644 index ef131fbbcd4..00000000000 --- a/poros/poros/lowering/fuse_meshgrid.cpp +++ /dev/null @@ -1,111 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file fuse_meshgrid.cpp -* @author Lin Xiao Chun (linxiaochun@baidu.com) -* @date 2022-04-29 14:56:48 -* @brief -**/ - -#include "poros/lowering/fuse_meshgrid.h" - -#include - -namespace baidu { -namespace mirana { -namespace poros { -/** - * rewrite meshgrid with `ones + transpose + mul` - * @param graph - * @return - */ -bool FuseMeshgrid::fuse(std::shared_ptr graph) { - if (try_to_find_meshgrid(graph->block())) { - std::string old_pattern = R"IR( - graph(%x_1 : Tensor, %y_1 : Tensor ): - %1 : Tensor[] = prim::ListConstruct(%x_1, %y_1) - %2 : Tensor[] = aten::meshgrid(%1) - return (%2))IR"; - - std::string new_pattern = R"IR( - graph(%x_1 : Tensor, %y_1 : Tensor): - %device.1 : Device = prim::device(%x_1) - %2 : NoneType = prim::Constant() - %3 : int = prim::Constant[value=1]() - %4 : int = prim::Constant[value=0]() - %5 : int[] = aten::size(%y_1) - %6 : int = aten::__getitem__(%5, %4) - %7 : int[] = prim::ListConstruct(%3, %6) - %x_dtype : int = prim::dtype(%x_1) - %8 : Tensor = aten::ones(%7, %x_dtype, %2, %device.1, %2) - %10 : Tensor = aten::unsqueeze(%x_1, %4) - %11 : Tensor = aten::transpose(%10, %4, %3) - %12 : Tensor = aten::mul(%8, %11) - - %25 : int[] = aten::size(%x_1) - %26 : int = aten::__getitem__(%25, %4) - %27 : int[] = prim::ListConstruct(%26, %3) - %y_dtype : int = prim::dtype(%y_1) - %28 : Tensor = aten::ones(%27, %y_dtype, %2, %device.1, %2) - %29 : Tensor = aten::unsqueeze(%y_1, %4) - %18 : Tensor = aten::mul(%28, %29) - - %19 : Tensor[] = prim::ListConstruct(%12, %18) - return (%19))IR"; - torch::jit::SubgraphRewriter std_rewriter; - std_rewriter.RegisterRewritePattern(old_pattern, new_pattern); - std_rewriter.runOnGraph(graph); - - return true; - } - return false; - -} - -FuseMeshgrid::FuseMeshgrid() = default; - -/** - * find out whether meshgrid exists - * @param block - * @return bool: true if meshgrid exists, else false - */ -bool FuseMeshgrid::try_to_find_meshgrid(torch::jit::Block *block) { - bool graph_changed = false; - auto it = block->nodes().begin(); - while (it != block->nodes().end()) { - auto node = *it; - ++it; //++it first, node may be destroyed later。 - for (auto sub_block: node->blocks()) { - if (try_to_find_meshgrid(sub_block)) { - graph_changed = true; - } - } - //只处理 aten::conv + batch_norm的场景 - if (node->kind() != torch::jit::aten::meshgrid) { - continue; - } - - record_transform(torch::jit::aten::meshgrid)->to(torch::jit::aten::ones, torch::jit::aten::unsqueeze, - torch::jit::aten::transpose, torch::jit::aten::mul); - graph_changed = true; - } - return graph_changed; -} - -REGISTER_OP_FUSER(FuseMeshgrid) - -} -} -}// namespace diff --git a/poros/poros/lowering/fuse_meshgrid.h b/poros/poros/lowering/fuse_meshgrid.h deleted file mode 100644 index d18dfd25d55..00000000000 --- a/poros/poros/lowering/fuse_meshgrid.h +++ /dev/null @@ -1,43 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file fuse_meshgrid.h -* @author Lin Xiao Chun (linxiaochun@baidu.com) -* @date 2022-04-29 14:56:57 -* @brief -**/ - -#pragma once - -#include "poros/lowering/op_fuse_pass.h" - -namespace baidu { -namespace mirana { -namespace poros { - -class FuseMeshgrid : public IFuser { -public: - FuseMeshgrid(); - - bool fuse(std::shared_ptr graph); - -private: - bool try_to_find_meshgrid(torch::jit::Block *block); - -}; - -} -} -} \ No newline at end of file diff --git a/poros/poros/lowering/input_param_propagate.cpp b/poros/poros/lowering/input_param_propagate.cpp deleted file mode 100644 index dda36c0170a..00000000000 --- a/poros/poros/lowering/input_param_propagate.cpp +++ /dev/null @@ -1,94 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file input_param_propagate.cpp -* @author huangben@baidu.com -* @date 2021-08-18 14:56:57 -* @brief -**/ -#include "poros/lowering/lowering_pass.h" - -#include -#include -#include -#include -#include -#include - -namespace baidu { -namespace mirana { -namespace poros { - -namespace { -using namespace torch::jit; -struct InputParamPropagate { - InputParamPropagate(std::shared_ptr graph) : graph_(std::move(graph)) {} - - void run(std::vector& stack_vec) { - if (stack_vec.size() == 0) { - return; - } - - auto check_input_param_unchanged = [](std::vector& stack_vec, size_t offset) { - if (stack_vec.size() == 1) { - return true; - } - auto ivalue = stack_vec[0][offset]; - for (size_t idx = 1; idx < stack_vec.size(); ++idx) { - if (stack_vec[idx][offset] != ivalue) { - return false; - } - } - return true; - }; - - auto g_inputs = graph_->inputs(); - size_t extra_offset = 0; - for (size_t offset = 0; offset < stack_vec[0].size(); ++offset) { - if (stack_vec[0][offset].isBool() || stack_vec[0][offset].isInt()) { - if (check_input_param_unchanged(stack_vec, offset)) { - WithInsertPoint guard(graph_->block()->nodes().front()); - auto insert_value = graph_->insertConstant(stack_vec[0][offset]); - if (g_inputs.size() == stack_vec[0].size()) { - g_inputs[offset]->replaceAllUsesWith(insert_value); - } else { - //TODO: this type check is not comprehensive. It may lead some bug with unexpected input data. - while (c10::ClassTypePtr c = g_inputs[offset + extra_offset]->type()->cast()) { - if (c->is_module()) { - extra_offset++; - } - } - g_inputs[offset + extra_offset]->replaceAllUsesWith(insert_value); - } - } - } - } - return; - } - -private: - std::shared_ptr graph_; -}; -} // namespace - -void input_param_propagate(std::shared_ptr graph, - std::vector>& stack_vec) { - InputParamPropagate ipp(std::move(graph)); - ipp.run(stack_vec); -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/lowering/link_mutable_list_pass.cpp b/poros/poros/lowering/link_mutable_list_pass.cpp deleted file mode 100644 index b7eed72f5bf..00000000000 --- a/poros/poros/lowering/link_mutable_list_pass.cpp +++ /dev/null @@ -1,204 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file link_mutable_list_pass.cpp -* @author tianshaoqing@baidu.com -* @date Thu May 9 11:15:49 CST 2022 -* @brief -**/ - -#include "poros/lowering/lowering_pass.h" - -#include -#include -#include -#include -#include - -#include "poros/context/poros_global.h" -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -namespace { -using namespace torch::jit; - -struct LinkMutableList { - LinkMutableList(std::shared_ptr graph) : graph_(std::move(graph)), - _mutable_list_ops_set(PorosGlobalContext::instance().supported_mutable_ops_set){} - - void run() { - GRAPH_DUMP("Before linking mutable list Graph: ", graph_); - bool changed = handle_mutable_list(graph_->block()); - if (changed) { - ConstantPropagation(graph_); - EliminateDeadCode(graph_); - EliminateCommonSubexpression(graph_); - ConstantPooling(graph_); - } - GRAPH_DUMP("After linking mutable list Graph: ", graph_); - return; - } - -private: - std::shared_ptr graph_; - - const std::set _mutable_list_ops_set; - // handle_mutable_list功能:将mutable的op输入与输出串联起来 - - // 通常(不含子block)情况: - // --------------------------------------------- - // %l1 : Tensor[] = aten::append(%list, %x1) - // %l2 : Tensor[] = aten::append(%list, %x2) - // %l3 : Tensor[] = aten::append(%list, %x3) - // %l4 : Tensor[] = aten::append(%list, %x4) - // --------------------------------------------- - // 转化为以下IR: - // --------------------------------------------- - // %l1 : Tensor[] = aten::append(%list, %x1) - // %l2 : Tensor[] = aten::append(%l1, %x2) - // %l3 : Tensor[] = aten::append(%l2, %x3) - // %l4 : Tensor[] = aten::append(%l3, %x4) - // --------------------------------------------- - - // 特殊(含子block)情况,以下面的IR为例: - // ---------------------------------------------- - // %list : Tensor[] = prim::ListConstruct(...) - // %l1 : Tensor[] = aten::append(%list, %x1) - // %l2 : Tensor[] = aten::append(%list, %x2) - // = prim::Loop(%5, %2) - // block0(%i : int): - // %l3 : Tensor[] = aten::_set_item(%list, %i, %x3) - // %l4 : Tensor[] = aten::append(%list, %x4) - // -> (%2) - // %l5 : Tensor[] = aten::append(%list, %x5) - // %%list2 : Tensor[] = aten::slice(%list, %4, %3, %4) - // ----------------------------------------------- - // 只对最外层主图中的mutable list op 输入输出串起来,而子block中不串。转化为以下IR: - // ----------------------------------------------- - // %list : Tensor[] = prim::ListConstruct(...) - // %l1 : Tensor[] = aten::append(%list, %x1) - // %l2 : Tensor[] = aten::append(%l1, %x2) - // = prim::Loop(%5, %2) - // block0(%i : int): - // %l3 : Tensor[] = aten::_set_item(%l2, %i, %x3) - // %l4 : Tensor[] = aten::append(%l2, %x4) - // -> (%2) - // %l5 : Tensor[] = aten::append(%l2, %x5) - // %%list2 : Tensor[] = aten::slice(%l5, %4, %3, %4) - // ---------------------------------------------- - // 只要保证子block中的mutable list op不合入子图就行,主图中子图的mutable可以不回传 - bool handle_mutable_list(Block* block) { - bool changed = false; - for (auto it = block->nodes().begin(); it != block->nodes().end(); ) { - Node* node = *it; - ++it; - if (_mutable_list_ops_set.find(node->kind()) != _mutable_list_ops_set.end()) { - if (node->outputs().size() == 1) { - changed = true; - node->input(0)->replaceAllUsesAfterNodeWith(node, node->output(0)); - } else { - LOG(WARNING) << "mutable op: " << node_info(node) << " output size() != 1. " << - "This situation is not yet supported."; - } - } - } - return changed; - } - // 以下是曾经实现的版本: - // 版本一,本应是最理想的版本,但是无跑通 - // ---------------------------------------------- - // %list : Tensor[] = prim::ListConstruct(...) - // %l1 : Tensor[] = aten::append(%list, %x1) - // %l2 : Tensor[] = aten::append(%l1, %x2) - // = prim::Loop(%5, %2) - // block0(%i : int): - // %l3 : Tensor[] = aten::_set_item(%l2, %i, %x3) <------执行到这步出错 - // %l4 : Tensor[] = aten::append(%l2, %x4) - // -> (%2) - // %l5 : Tensor[] = aten::append(%l2, %x5) - // %%list2 : Tensor[] = aten::slice(%l2, %4, %3, %4) - // ------------------------------------------------- - // *在子block中对某value使用replaceAllUsesAfterNodeWith时,如果block外面也有value的user的话jit会出错 - // 本例子中,由于l2在子block外部也有users,在给子block以外的%l2替换成%l3时会出错 - /* - bool handle_mutable_list(Block* block) { - bool changed = false; - for (auto it = block->nodes().begin(); it != block->nodes().end(); ) { - Node* node = *it; - ++it; - for (Block* subblock : node->blocks()) { - changed |= handle_mutable_list(subblock); - } - if (_mutable_list_ops_set.find(node->kind()) != _mutable_list_ops_set.end()) { - changed = true; - node->input(0)->replaceAllUsesAfterNodeWith(node, node->output(0)); - } - } - return changed; - }*/ - - // 版本二,只替换同一block下且在本node之后的node(mutable) - // 执行后: - // ---------------------------------------------- - // %list : Tensor[] = prim::ListConstruct(...) - // %l1 : Tensor[] = aten::append(%list, %x1) - // %l2 : Tensor[] = aten::append(%l1, %x2) - // = prim::Loop(%5, %2) - // block0(%i : int): - // %l3 : Tensor[] = aten::_set_item(%list, %i, %x3) - // %l4 : Tensor[] = aten::append(%l3, %x4) - // -> (%2) - // %l5 : Tensor[] = aten::append(%l2, %x5) - // %%list2 : Tensor[] = aten::slice(%l5, %4, %3, %4) - // ------------------------------------------------ - // 作用域只能在自己node归属的block中,导致子block调用了最开始的mutable(%list), - // 主图中子图的mutable需要回传,子block中子图的mutable需要回传,依赖回传。 - /* - bool handle_mutable_list(Block* block) { - bool changed = false; - for (auto it = block->nodes().begin(); it != block->nodes().end(); ) { - Node* node = *it; - ++it; - for (Block* subblock : node->blocks()) { - changed |= handle_mutable_list(subblock); - } - if (_mutable_list_ops_set.find(node->kind()) != _mutable_list_ops_set.end()) { - changed = true; - torch::jit::use_list use_list = node->input(0)->uses(); - for (size_t u = 0; u < use_list.size(); u++) { - // 只替换同一block下且在本node之后的node - if (use_list[u].user->owningBlock() == block && use_list[u].user->isAfter(node)) { - use_list[u].user->replaceInput(use_list[u].offset, node->output(0)); - } - } - } - } - return changed; - }*/ -}; - -} // namespace - -void link_mutable_list(std::shared_ptr graph) { - LinkMutableList lml(std::move(graph)); - lml.run(); -} - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/lowering/lowering_pass.h b/poros/poros/lowering/lowering_pass.h deleted file mode 100644 index e7b8721f6e9..00000000000 --- a/poros/poros/lowering/lowering_pass.h +++ /dev/null @@ -1,190 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file lowering_pass.h -* @author Lin Xiao Chun (linxiaochun@baidu.com) -* @date 2022-03-31 16:11:18 -* @brief -**/ -#pragma once - -#include - -#include -#include - -namespace baidu { -namespace mirana { -namespace poros { - -/** -* @brief 删除graph中,纯粹的prim::RaiseException分支。 -* (否则graph会被分割成过多的block) -**/ -void eliminate_exception_pass(std::shared_ptr graph); - -/** -* @brief 替换graph中,prim::ListConstruct 类型的节点后面,紧跟 prim::ListUnpack 类型的情况。 -* prim::ListConstruct 用来将多个元素构建成list -* prim::ListUnpack 用来将一个list打散成多个元素。 -* 当这两个节点处理同一个list,且节点间没有其他可能改变该list的情况时,将这两个节点抵消。 -**/ -void eliminate_some_list(std::shared_ptr graph); - -/** -* @brief 替换graph中,prim::DictConstruct 类型的节点后面,跟的全部是 aten::__getitem__ 类型的情况(且dict的key是常量)。 -* prim::DictConstruct 用来将多个元素构建成dict -* aten::__getitem__ 用来从list 或者 dict 中获取元素。 -* 当DictConstruct生成的dict,只被aten::__getitem__ 调用,且没有其他可能改变该dict的情况时,将这两类op抵消。 -**/ -void eliminate_some_dict(std::shared_ptr graph); - -/** -* @brief 删除graph中,未被使用的aten::copy_节点。 -* -**/ -void eliminate_useless_copy(std::shared_ptr graph); - - -/** - * @brief 尝试用maxpool 代替 maxpool_with_indeces. - * 以 maxpoll2d 为例: - * maxpoll2d_with_indices 的schema为:aten::max_pool2d_with_indices(Tensor self, int[2] kernel_size, int[2] stride=[], int[2] padding=0, int[2] dilation=1, bool ceil_mode=False) -> (Tensor, Tensor) - * 而 maxpoll 的schema为:aten::max_pool2d(Tensor self, int[2] kernel_size, int[2] stride=[], int[2] padding=0, int[2] dilation=1, bool ceil_mode=False) -> Tensor - * 这两个op,输入参数完全一致,输出上,max_pool2d_with_indices有两个输出,第一个输出与max_pool2d的输出完全一致,第二个输出为indeces信息。 - * 当 max_pool2d_with_indices 的第二个输出indices,后续没有其他op使用该value的时候, - * 我们直接用max_pool2d 替代 max_pool2d_with_indices。 - **/ -void eliminate_maxpool_with_indices(std::shared_ptr graph); - -/** -* @brief 将符合条件的loop进行循环展开,避免过多的block,影响子图分割的逻辑。 -* 本function很大程度上参考了jit原生的UnrollLoop的实现, -* 考虑到原生的UnrollLoop支持的bodysize和loopcount不符合poros的预期,且原生实现不提供修改参数的接口。 -* 故重新实现该function, 调整loop展开的条件和部分细节。 -**/ -void unrolling_loop(std::shared_ptr graph); - -/** -* @brief 替换graph中,aten::std 的算子为 aten::var + aten::sqrt。 -* 依据:标准差(aten::std) = 方差(aten::var) 的算术平方根(aten::sqrt) -**/ -void unpack_std(std::shared_ptr& graph); - -/** -* @brief 替换graph中,aten::var 的算子为 aten::mul + aten::mean 等。 -* 参考pytorch-1.9.0 中该算子的实现: https://github.com/pytorch/pytorch/blob/v1.9.0/aten/src/ATen/native/ReduceOps.cpp#L1380 -**/ -void unpack_var(std::shared_ptr& graph); - -/** -* @brief 尝试将aten::percentFormat 的结果变成常量。 -* 背景: aten::percentFormat 的功能主要是用于字符串的组装, -* 且常常配合 prim::If 这个op进行字符串的比较,实现分支选择。 -* 当precentFormat 的输入都是常量的时候,尝试直接计算出这个算子的结果,替换成常量 -* 进一步配合prim::If 的条件判断是否为常量,最终配合达到删除不必要的分支的目的。 -**/ -void freeze_percentformat(std::shared_ptr graph); - -/** -* @brief 尝试固定aten::size的结果。需要配合后续aten::size的输出的使用进行判断。 -* 注意: 本function必须在预热数据处理完整个graph之后再使用,且依赖于预热数据覆盖的全面程度。 -**/ -void freeze_aten_size(std::shared_ptr graph); - -/** -* @brief 尝试固定aten::len的结果。需要配合后续aten::len的输出的使用进行判断。 -* 注意: 本function必须在预热数据处理完整个graph之后再使用,且依赖于预热数据覆盖的全面程度。 -**/ -void freeze_aten_len(std::shared_ptr graph); - -/** -* @brief 尝试固定aten::dim的结果。需要配合后续aten::dim的输出的使用进行判断。 -* 注意: 本function必须在预热数据处理完整个graph之后再使用,且依赖于预热数据覆盖的全面程度。 -**/ -void freeze_aten_dim(std::shared_ptr graph); - -/** -* @brief 当遇到使用ListConstruct对1个constant进行append时,可以讲output直接替换为1个constant -* 例如:float 替换为 (float, float,..) -**/ -void freeze_list_construct(std::shared_ptr graph); - -/** -* @brief 针对graph的简单类型输入【bool or int 类型】,尝试进行剪枝。 -* 当多轮预热数据的简单类型输入保持不变,则认为该输入可以用常量进行替代。 -* 注意: 本function必须在预热数据处理完整个graph之后再使用,且依赖于预热数据覆盖的全面程度。 -**/ -void input_param_propagate(std::shared_ptr graph, - std::vector>& stack_vec); - -/** -* @brief 移除graph中,跟踪bool类型的prim::profile节点。 -* 这些prim::profile在数据预热阶段(IvalueAnalysis)添加进graph,数据预热完成后,需要相应移除这些节点。 -**/ -void remove_simple_type_profile_nodes(std::shared_ptr graph); - -/** -* @brief 使用log(softmax())替代log_softmax() -**/ -void replace_log_softmax(std::shared_ptr graph); - -/** -* @brief 使用log(sigmoid())替代log_sigmoid() -**/ -void replace_log_sigmoid(std::shared_ptr graph); - -/** -* @brief 将包含list类型引用语义的op输入与输出串联起来。 -**/ -void link_mutable_list(std::shared_ptr graph); - -/** -* @brief 直接删除图中与infer无关的节点。 -**/ -void eliminate_simple_useless_nodes(std::shared_ptr graph); - -/** - * @brief 删除子图内部或输入相关的无用节点。(当前支持aten::to.device,aten::contiguous,aten::dropout和aten::detach) - * 注意:1、删除子图内部节点(is_input == false)必须在拷贝的子图上,否则fallback会出错。 - * 2、删除(替换)子图输入节点(is_input == true)必须在子图转engine成功后。 - * - * @param [in] subgraph : 要删无用节点的子图 - * @param [in] subgraph_node : subgraph对应的子图节点,类型必须是prim::CudaFusionGroup - * @param [in] is_input : true表示要删除的是子图输入的节点,false表示删除子图内部节点 - * - * @return bool - * @retval true => 删除节点成功 false => 如果删完无用节点后的子图node数量为0,返回false unmerge -**/ -bool eliminate_subgraph_useless_nodes(std::shared_ptr subgraph, - torch::jit::Node& subgraph_node, - const bool is_input); - -/** -* @brief 检查并替换有问题的constant -**/ -void replace_illegal_constant(std::shared_ptr graph); - -/** -* @brief 替换aten::pad。该op可视做多种pad的集合,只是用mode来设置pad方式(包括:constant、reflect、replicate还有circular)。 -* mode == constant时,可替换为aten::constant_pad_nd,已实现。 -* todo: -* mode == refect时,替换为aten::reflection_pad -* mode == replicate时,替换为aten::replication_pad -**/ -void replace_pad(std::shared_ptr graph); -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/lowering/op_fuse_pass.cpp b/poros/poros/lowering/op_fuse_pass.cpp deleted file mode 100644 index d6eb56ec816..00000000000 --- a/poros/poros/lowering/op_fuse_pass.cpp +++ /dev/null @@ -1,139 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file op_fuse_pass.cpp -* @author Lin Xiao Chun (linxiaochun@baidu.com) -* @date 2022-03-31 16:11:18 -* @brief -**/ -#include "poros/lowering/op_fuse_pass.h" - -#include -#include -#include -#include -#include - -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -std::string IFuser::info() { - std::string info = "OP Fuser:" + IFuser::name_ + " "; - - for (auto ori: IFuser::fused_ops) { - info += "[" + ori->info() + "]"; - } - return info; - -} - -void IFuser::setName(const std::string name) { - IFuser::name_ = name; -} - -void IFuser::reset() { - IFuser::fused_ops.clear(); -} - -std::string FusedOpsRecord::info() { - std::string info; - for (auto it = from_ops_.begin(); it != from_ops_.end(); it++) { - info += std::string(it->toUnqualString()); - if (it != from_ops_.end() - 1) { - info += ","; - } - } - info += " => "; - for (auto it = to_ops_.begin(); it != to_ops_.end(); it++) { - info += std::string(it->toUnqualString()); - if (it != to_ops_.end() - 1) { - info += ","; - } - } - return info; -} - -void FusedOpsRecord::from() { - -} - -void FusedOpsRecord::to() { - -} - -void fuse_ops_preprocess(std::shared_ptr graph) { - IFuserManager *manager = IFuserManager::get_instance(); - manager->preprocess_fuse(std::move(graph)); - -} - -void fuse_ops_prewarm(std::shared_ptr graph) { - IFuserManager *manager = IFuserManager::get_instance(); - manager->prewarm_fuse(std::move(graph)); - -} - -IFuserManager *IFuserManager::get_instance() { - static IFuserManager manager; - return &manager; -} - -std::string IFuserManager::register_fuser(const std::shared_ptr &fuser, const std::string &name) { - fuser->setName(name); - preprocess_fusers.push_back(fuser); - prewarm_fusers.push_back(fuser); - return name; -} - -void IFuserManager::preprocess_fuse(std::shared_ptr graph) { - bool graph_changed = false; - for (auto &&fuser: preprocess_fusers) { - fuser->reset(); - if (fuser->fuse(graph)) { - LOG(INFO) << fuser->info(); - graph_changed = true; - } - } - if (graph_changed) { - ConstantPropagation(graph); - EliminateDeadCode(graph); - EliminateCommonSubexpression(graph); - ConstantPooling(graph); - } -} - -void IFuserManager::prewarm_fuse(std::shared_ptr graph) { - bool graph_changed = false; - for (auto &&fuser: prewarm_fusers) { - fuser->reset(); - if (fuser->fuse(graph)) { - LOG(INFO) << fuser->info(); - graph_changed = true; - } - } - if (graph_changed) { - ConstantPropagation(graph); - EliminateDeadCode(graph); - EliminateCommonSubexpression(graph); - ConstantPooling(graph); - } -} - -} -} -} \ No newline at end of file diff --git a/poros/poros/lowering/op_fuse_pass.h b/poros/poros/lowering/op_fuse_pass.h deleted file mode 100644 index 4b63cd6f949..00000000000 --- a/poros/poros/lowering/op_fuse_pass.h +++ /dev/null @@ -1,174 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file op_fuse_pass.h -* @author Lin Xiao Chun (linxiaochun@baidu.com) -* @date 2022-03-31 16:11:18 -* @brief -**/ - -#pragma once - -#include -#include -#include - -#include - -namespace baidu { -namespace mirana { -namespace poros { - -/** - * FusedOpsRecord - * only used for recording fusion infomation, DO NOT affect actual fusing logic - */ -class FusedOpsRecord { -public: - void from(); - - template - void from(torch::jit::Node *first, Rest ... rest); - - template - void from(torch::jit::NodeKind first, Rest ... rest); - - void to(); - - template - void to(torch::jit::Node *first, Rest ... rest); - - template - void to(torch::jit::NodeKind first, Rest ... rest); - - std::string info(); - -private: - - std::vector from_ops_; - std::vector to_ops_; -}; - -template -void FusedOpsRecord::from( torch::jit::Node *first, Rest ... rest) { - from_ops_.push_back(first->kind()); - from(rest...); // recursive call using pack expansion syntax -} - -template -void FusedOpsRecord::from( torch::jit::NodeKind first, Rest ... rest) { - from_ops_.push_back(first); - from(rest...); // recursive call using pack expansion syntax -} - -template -void FusedOpsRecord::to(torch::jit::Node *first, Rest ... rest) { - to_ops_.push_back(first->kind()); - to(rest...); -} - -template -void FusedOpsRecord::to(torch::jit::NodeKind first, Rest ... rest) { - to_ops_.push_back(first); - to(rest...); // recursive call using pack expansion syntax -} - -/** - * IFuser - * base class of all fusers - */ -class IFuser { -public: - IFuser() = default;; - - virtual ~IFuser() = default;; - - virtual bool fuse(std::shared_ptr graph) = 0; - - std::string info(); - - void reset(); - - - void setName(const std::string name); - - template - std::shared_ptr record_transform(First first, Rest ...rest); - -private: - std::vector> fused_ops; - - std::string name_; - -}; - -template -std::shared_ptr IFuser::record_transform(First first, Rest ... rest) { - auto f = std::make_shared(); - f->from(first, rest...); // recursive call using pack expansion syntax - fused_ops.push_back(f); - return f; -} - -/** - * IFuserManager - * manage the registration and application of fusers - */ -class IFuserManager { -public: - - static IFuserManager *get_instance(); - - std::string register_fuser(const std::shared_ptr &fuser, const std::string &name); - - /** - * apply all fusers in preprocess_fusers - * @param graph - */ - void preprocess_fuse(std::shared_ptr graph); - - /** - * apply all fusers in prewarm_fusers - * @param graph - */ - void prewarm_fuse(std::shared_ptr graph); - -private: - std::vector> preprocess_fusers; - std::vector> prewarm_fusers; - -}; - -/** - * trying to fuse ops during preprocessing stage - * @param graph - */ -void fuse_ops_preprocess(std::shared_ptr graph); - -/** - * trying to fuse ops during pre-warming stage, now it's same to fuse_ops_preprocess. - * @param graph - */ -void fuse_ops_prewarm(std::shared_ptr graph); - -#define REGISTER_OP_FUSER(T) \ - const std::string _G_NAME = []() -> std::string { \ - return IFuserManager::get_instance()->register_fuser( \ - std::make_shared(), #T); \ - }(); - -} -} -} diff --git a/poros/poros/lowering/remove_simple_type_profile_nodes.cpp b/poros/poros/lowering/remove_simple_type_profile_nodes.cpp deleted file mode 100644 index 78d08f87054..00000000000 --- a/poros/poros/lowering/remove_simple_type_profile_nodes.cpp +++ /dev/null @@ -1,89 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file remobe_simple_type_profile_nodes.cpp -* @author tianjinjin@baidu.com -* @date Mon May 10 11:06:53 CST 2021 -* @brief -**/ -#include "poros/lowering/lowering_pass.h" - -#include -#include -#include -#include -#include - -namespace baidu { -namespace mirana { -namespace poros { - -namespace { -using namespace torch::jit; - -struct RemoveSimpleTypeProfileNodes { - RemoveSimpleTypeProfileNodes(std::shared_ptr graph) : graph_(std::move(graph)) {} - void run() { - remove_profile_nodes(graph_->block()); - } - -private: - bool profiled_with_different_types(Value* v) { - std::vector types; - for (const auto& use : v->uses()) { - if (use.user->kind() == prim::profile) { - types.push_back(use.user->ty(attr::profiled_type)); - } - } - for (size_t i = 1; i < types.size(); ++i) { - if (types.at(i - 1) != types.at(i)) { - return true; - } - } - return false; - } - - bool is_simple_type_profile_node(Node* node) { - return node->ty(attr::profiled_type) != TensorType::get(); - } - - void remove_profile_nodes(Block* block) { - for (auto itr = block->nodes().begin(); itr != block->nodes().end(); itr++) { - if (itr->kind() == prim::profile && is_simple_type_profile_node(*itr)) { //todo - itr->output()->replaceAllUsesWith(itr->input()); - if (!profiled_with_different_types(itr->input())) { - itr->input()->setType(itr->ty(attr::profiled_type)); - } - itr.destroyCurrent(); - } else { - for (Block* ib : itr->blocks()) { - remove_profile_nodes(ib); - } - } - } - } - - std::shared_ptr graph_; -}; - -} // namespace - -void remove_simple_type_profile_nodes(std::shared_ptr graph) { - RemoveSimpleTypeProfileNodes(graph).run(); -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/lowering/replace_illegal_constant.cpp b/poros/poros/lowering/replace_illegal_constant.cpp deleted file mode 100644 index fe3eeecf52a..00000000000 --- a/poros/poros/lowering/replace_illegal_constant.cpp +++ /dev/null @@ -1,118 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file replace_illegal_constant.cpp -* @author tianshaoqing@baidu.com -* @date 2022-06-01 19:34:40 -* @brief -**/ -#include "poros/lowering/lowering_pass.h" - -#include -#include -#include -#include -#include - -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -namespace { -using namespace torch::jit; - -struct ReplaceIllegalConstant { - ReplaceIllegalConstant(std::shared_ptr graph) : graph_(std::move(graph)) {} - - void run() { - GRAPH_DUMP("before replace_illegal_constant Graph: ", graph_); - bool changed = find_and_replace_illegal_constant(graph_->block()); - if (changed) { - ConstantPropagation(graph_); - EliminateDeadCode(graph_); - EliminateCommonSubexpression(graph_); - ConstantPooling(graph_); - } - GRAPH_DUMP("after replace_illegal_constant Graph: ", graph_); - return; - } - -private: - bool check_constant_is_illegal(Node* node) { - bool is_illegal = false; - // 检查constant输出是否是int - if (node->kind() == torch::jit::prim::Constant && node->outputs().size() > 0 && - node->output(0)->type()->kind() == c10::TypeKind::IntType) { - torch::jit::IValue const_value = toIValue(node->output(0)); - // 这里的toInt返回的是int64_t - long const_double = const_value.toInt(); - // 判断int是否等于INT64_MAX,且有users - if ((const_double == INT64_MAX) && node->output(0)->hasUses()) { - is_illegal = true; - auto const_node_users = node->output(0)->uses(); - // 判断逻辑,目前只遇到了slice end输入为非法constant的情况,其他情况遇到再加 - for (size_t u = 0; u < const_node_users.size(); u++) { - if (const_node_users[u].user->kind() != torch::jit::aten::slice) { - is_illegal = false; - break; - } - } - } - } - return is_illegal; - } - - bool find_and_replace_illegal_constant(Block* block) { - bool graph_changed = false; - auto it = block->nodes().begin(); - while (it != block->nodes().end()) { - auto node = *it; - ++it; //++it first, node may be destroyed later。 - for (auto sub_block: node->blocks()) { - if (find_and_replace_illegal_constant(sub_block)) { - graph_changed = true; - } - } - - if (node->kind() == torch::jit::prim::Constant && - check_constant_is_illegal(node)) { - // 将slice end输入替换为none - torch::jit::Node* none_node = graph_->createNone(); - none_node->insertBefore(node); - node->output(0)->replaceAllUsesAfterNodeWith(node, none_node->output(0)); - LOG(INFO) << "Found illegal constant INT64_MAX used as index by aten::slice. Replace it with Constant None."; - node->destroy(); - graph_changed = true; - } - } - return graph_changed; - } - - std::shared_ptr graph_; - std::unordered_set useless_schema_set_; -}; - -} // namespace - -void replace_illegal_constant(std::shared_ptr graph) { - ReplaceIllegalConstant ric(std::move(graph)); - ric.run(); -} - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/lowering/replace_pad.cpp b/poros/poros/lowering/replace_pad.cpp deleted file mode 100644 index 4245dd017cc..00000000000 --- a/poros/poros/lowering/replace_pad.cpp +++ /dev/null @@ -1,112 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file replace_pad.cpp -* @author tianshaoqing@baidu.com -* @date 2022-11-09 19:34:40 -* @brief -**/ -#include "poros/lowering/lowering_pass.h" - -#include -#include -#include -#include -#include - -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -namespace { -using namespace torch::jit; - -struct ReplacePad { - ReplacePad(std::shared_ptr graph) : graph_(std::move(graph)) {} - - void run() { - GRAPH_DUMP("before replace pad Graph: ", graph_); - bool changed = find_and_replace_pad(graph_->block()); - if (changed) { - ConstantPropagation(graph_); - EliminateDeadCode(graph_); - EliminateCommonSubexpression(graph_); - ConstantPooling(graph_); - } - GRAPH_DUMP("after replace pad Graph: ", graph_); - return; - } - -private: - bool check_pad_constant_mode(Node* node) { - bool is_constant_mode = false; - // 检查schema是否为 - // aten::pad(Tensor self, int[] pad, str mode="constant", float? value=None) -> (Tensor) - if (node->kind() == c10::Symbol::fromQualString("aten::pad") && - node->inputs().size() == 4 && - node->input(1)->type()->isSubtypeOf(c10::ListType::ofInts()) && - node->input(2)->type()->isSubtypeOf(c10::StringType::get())) { - std::string pad_mode = toIValue(node->input(2)).value().toStringRef(); - if (pad_mode == "constant") { - is_constant_mode = true; - } - } - return is_constant_mode; - } - - bool find_and_replace_pad(Block* block) { - bool graph_changed = false; - auto it = block->nodes().begin(); - while (it != block->nodes().end()) { - auto node = *it; - ++it; //++it first, node may be destroyed later。 - for (auto sub_block: node->blocks()) { - if (find_and_replace_pad(sub_block)) { - graph_changed = true; - } - } - // replace aten::pad with aten::constant_pad_nd when its padding mode is "constant". - if (node->kind() == c10::Symbol::fromQualString("aten::pad") && - check_pad_constant_mode(node)) { - torch::jit::Node* constant_pad_nd_node = graph_->create(torch::jit::aten::constant_pad_nd); - constant_pad_nd_node->addInput(node->input(0)); - constant_pad_nd_node->addInput(node->input(1)); - constant_pad_nd_node->addInput(node->input(3)); - constant_pad_nd_node->insertBefore(node); - node->output(0)->replaceAllUsesAfterNodeWith(node, constant_pad_nd_node->output(0)); - LOG(INFO) << "Replace aten::pad which padding mode is \"constant\" with aten::constant_pad_nd."; - node->destroy(); - graph_changed = true; - } - } - return graph_changed; - } - - std::shared_ptr graph_; - std::unordered_set useless_schema_set_; -}; - -} // namespace - -void replace_pad(std::shared_ptr graph) { - ReplacePad rp(std::move(graph)); - rp.run(); -} - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/lowering/segment_post_processing.cpp b/poros/poros/lowering/segment_post_processing.cpp deleted file mode 100644 index 88beb3efd94..00000000000 --- a/poros/poros/lowering/segment_post_processing.cpp +++ /dev/null @@ -1,72 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file segment_post_processing.cpp -* @author tianshaoqing@baidu.com -* @date Thu May 27 11:13:02 CST 2022 -* @brief -**/ - -#include "poros/lowering/segment_post_processing.h" - -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -using namespace torch::jit; - -void subgraph_outputs_int2long(torch::jit::Graph* parent_graph, - torch::jit::Node& subgraph_node, - std::shared_ptr subgraph) { - AT_ASSERT(subgraph_node.kind() == torch::jit::prim::CudaFusionGroup); - // 检查子图的每个Tensor的输出类型 - for (size_t i = 0; i < subgraph->outputs().size(); i++) { - torch::jit::Value* output_value = subgraph->outputs()[i]; - if (output_value->type()->isSubtypeOf(c10::TensorType::get())) { - auto subgraph_output_type = output_value->type()->cast(); - if (subgraph_output_type->scalarType() == at::ScalarType::Long) { - // 如果子图Tensor输出是Long,则添加aten::to.dtype,schema如下: - // aten::to.dtype(Tensor self, ScalarType dtype, bool non_blocking=False, bool copy=False, MemoryFormat? memory_format=None) -> Tensor - LOG(INFO) << "Find output type is Long, which is " << node_info(&subgraph_node) << " output[" << i << - "] %" << output_value->debugName() << ". Add aten::to(Long) node."; - torch::jit::Node* to_long_node = parent_graph->create(torch::jit::aten::to, 1); - to_long_node->insertAfter(&subgraph_node); - to_long_node->addInput(subgraph_node.output(i)); - // 不用setInsertPoint的话默认将constant插入到图的末尾,movebefore将constant移到to_long_node之前 - torch::jit::Value* false_value = parent_graph->insertConstant(false); - false_value->node()->moveBefore(to_long_node); - torch::jit::Value* type_value = parent_graph->insertConstant(c10::ScalarType::Long); - type_value->node()->moveBefore(to_long_node); - torch::jit::Node* none_node = parent_graph->createNone(); - none_node->insertBefore(to_long_node); - - to_long_node->addInput(type_value); - to_long_node->addInput(false_value); - to_long_node->addInput(false_value); - to_long_node->addInput(none_node->output(0)); - - // must set output type - to_long_node->output(0)->setType(subgraph_output_type); - subgraph_node.output(i)->replaceAllUsesAfterNodeWith(to_long_node, to_long_node->output(0)); - } - } - } -}; - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/lowering/segment_post_processing.h b/poros/poros/lowering/segment_post_processing.h deleted file mode 100644 index d1b71332100..00000000000 --- a/poros/poros/lowering/segment_post_processing.h +++ /dev/null @@ -1,48 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file segment_post_processing.h -* @author tianshaoqing@baidu.com -* @date 2022-05-27 11:11:18 -* @brief -**/ - -#pragma once - -#include - -namespace baidu { -namespace mirana { -namespace poros { - -/** - * @brief 有的子图输出的tensor原本是long类型,但是trtengine只支持int类型。 - * 那么就需要在engine后添加aten::to(long)的操作将其还原回去。 - * 避免有的op会强制检查long类型(例如:aten::index) - * - * @param [in] parent_graph : subgraph_node的owning_graph - * @param [in] subgraph_node : 子图节点,类型必须是prim::CudaFusionGroup - * @param [in] subgraph : 子图节点所对应的子图 - * - * @return - * @retval -**/ -void subgraph_outputs_int2long(torch::jit::Graph* parent_graph, - torch::jit::Node& subgraph_node, - std::shared_ptr subgraph); - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/lowering/try_to_freeze_aten_dim.cpp b/poros/poros/lowering/try_to_freeze_aten_dim.cpp deleted file mode 100644 index a77592095d7..00000000000 --- a/poros/poros/lowering/try_to_freeze_aten_dim.cpp +++ /dev/null @@ -1,103 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file: /poros/baidu/mirana/poros/src/poros/lowering/try_to_freeze_aten_dim.cpp -* @author: zhangfan51@baidu.com -* @data: 2022-03-24 16:02:50 -* @brief: -**/ - -#include "poros/lowering/lowering_pass.h" - -#include -#include - -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -namespace { -using namespace torch::jit; - -bool has_type_and_dim(const Value* value) { - auto op = value->type()->cast(); - return op->sizes().size().has_value() && op->scalarType().has_value(); -} - -struct FreezeAtenDim { - FreezeAtenDim(std::shared_ptr graph) : graph_(std::move(graph)) {} - - void run() { - try_to_freeze_aten_dim(graph_->block()); - // 运行一遍常量折叠 - torch::jit::ConstantPropagation(graph_); - // torch::jit::ConstantPooling(graph_); - } - -private: - void replace_aten_dim(Node* node, int inplace_number) { - LOG(INFO) << "try to replace the output of node :" << node_info(node) - << " with constant value " << inplace_number; - torch::jit::WithInsertPoint guard(graph_->block()->nodes().front()); - auto len_const = graph_->insertConstant(inplace_number); - node->outputs().at(0)->replaceAllUsesWith(len_const); - } - - /* - * @brief 尝试将aten::dim的返回值变成常量,如果 aten::dim的input的Tensor经过预热后维度确定,则可以使用常量代替 - **/ - void try_to_freeze_aten_dim(Block* block) { - auto it = block->nodes().begin(); - while (it != block->nodes().end()) { - auto node = *it; - ++it; //先++it, node 可能destroy掉。 - for (auto block : node->blocks()) { - try_to_freeze_aten_dim(block); - } - - //只handle aten::dim 的场景 - if (node->kind() != aten::dim) { - continue; - } - - //输入是tensor,并且该tensor包含shape信息 - if ((node->inputs()[0])->type()->isSubtypeOf(c10::TensorType::get()) && - has_type_and_dim(node->inputs()[0])) { - auto sizes = (node->inputs()[0])->type()->cast()->sizes(); - // if (sizes[0].has_value()) { - auto ndim = sizes.size(); - replace_aten_dim(node, *ndim); - // } - continue; - } - } - } - - std::shared_ptr graph_; -}; - -} // namespace - -void freeze_aten_dim(std::shared_ptr graph) { - LOG(INFO) << "Running poros freeze_aten_len passes"; - FreezeAtenDim pss(std::move(graph)); - pss.run(); -} - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/lowering/try_to_freeze_aten_len.cpp b/poros/poros/lowering/try_to_freeze_aten_len.cpp deleted file mode 100644 index d63fea3a769..00000000000 --- a/poros/poros/lowering/try_to_freeze_aten_len.cpp +++ /dev/null @@ -1,167 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file try_to_freeze_aten_len.cpp -* @author tianjinjin@baidu.com -* @date Sun Sep 26 20:00:01 CST 2021 -* @brief -**/ - -#include "poros/lowering/lowering_pass.h" - -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -namespace { -using namespace torch::jit; - -bool has_type_and_dim(const Value* value) { - auto op = value->type()->cast(); - return op->sizes().size().has_value() && op->scalarType().has_value(); -} - -struct FreezeAtenLen { - FreezeAtenLen(std::shared_ptr graph) : graph_(std::move(graph)) {} - - void run() { - try_to_freeze_aten_len(graph_->block()); - } - -private: - void replace_aten_len(Node* node, int inplace_number) { - LOG(INFO) << "try to replace the output of node :" << node_info(node) - << " with constant value " << inplace_number; - torch::jit::WithInsertPoint guard(graph_->block()->nodes().front()); - auto len_const = graph_->insertConstant(inplace_number); - node->outputs().at(0)->replaceAllUsesWith(len_const); - } - - bool try_to_replace_listconstruct_len(Node* node, Node* len_node) { - if (node->kind() != prim::ListConstruct) { - return false; - } - for (auto &use : node->outputs()[0]->uses()) { - if (use.user->owningBlock() != node->owningBlock() || - use.user->kind() == aten::append) { - return false; - } - } - replace_aten_len(len_node, node->inputs().size()); - return true; - } - - /** - * @brief 尝试将aten::len的返回值变成常量,当前支持的场景: - * 1. aten::len 的输入是一个tensor(前提是我们认为tensor的size可能是dynamic的,但是len是确定的) - * 2. aten::len 的输入是prim::ListConstruct构建的list,当list的长度可以明确的时候,进行常量替换。 - * 3. aten::len 的输入是aten::unbind,根据该算子的语义,获取其len并替换。 - * 4. aten::len 的输入是aten::meshgrid,由于这类算子不改变输入的len信息,进一步获取算子的输入,尝试进行常量替换。 - **/ - void try_to_freeze_aten_len(Block* block) { - auto it = block->nodes().begin(); - while (it != block->nodes().end()) { - auto node = *it; - ++it; //先++it, node 可能destroy掉。 - for (auto block : node->blocks()) { - try_to_freeze_aten_len(block); - } - - //只handle aten::len 的场景 - if (node->kind() != aten::len) { - continue; - } - - //输入是一个tensor的场景, aten::len的结果 - if ((node->inputs()[0])->type()->isSubtypeOf(c10::TensorType::get()) && - has_type_and_dim(node->inputs()[0])) { - LOG(INFO) << "input is tensor situation."; - auto sizes = (node->inputs()[0])->type()->cast()->sizes(); - if (sizes[0].has_value()) { - int len = sizes[0].value(); - replace_aten_len(node, len); - } - continue; - // std::vector dims; - // if (gen_dims_for_tensor(node->inputs()[0], dims)) { - // int len = (dims.size()) & INT_MAX; - // replace_aten_len(node, len); - // } - // continue; - } - - //输入非tensor的场景,根据输入类型节点的类型简单判断。 - auto input_node = (node->inputs()[0])->node(); - switch (input_node->kind()) { - // unbind: 等于第一个输入的 - case aten::unbind: { - LOG(INFO) << "input is produced by aten::unbind situation."; - if (has_type_and_dim(input_node->inputs()[0])) { - std::vector dims; - if (gen_dims_for_tensor(input_node->inputs()[0], dims) && - input_node->inputs()[1]->node()->kind() == prim::Constant) { - int dim = toIValue(input_node->inputs()[1]->node()->output()).value().toInt(); - dim = dim < 0 ? dims.size() + dim : dim; - //非dynamic的维度 - if (dims[dim] != -1) { - auto len = dims[dim]; - replace_aten_len(node, len); - torch::jit::WithInsertPoint guard(graph_->block()->nodes().front()); - auto len_const = graph_->insertConstant(len); - node->outputs().at(0)->replaceAllUsesWith(len_const); - } - } - } - break; - } - //这些op的输入输出的len不会发生变化。再找一下这类op的输入。 - case aten::meshgrid: { - LOG(INFO) << "input is produced by aten:meshgrid situation."; - if ((input_node->inputs()[0])->node()->kind() == prim::ListConstruct) { - try_to_replace_listconstruct_len((input_node->inputs()[0])->node(), node); - } - break; - } - //prim::ListConstruct 的情况 - case prim::ListConstruct: { - LOG(INFO) << "input is produced by prim::ListConstruct situation."; - try_to_replace_listconstruct_len(input_node, node); - break; - } - default: { - //遇到目前不支持的类型,直接返回,不做处理。 - LOG(INFO) << "unsupported situation. input_node is: " << node_info(input_node); - break; - } - } - } - } - - std::shared_ptr graph_; -}; - -} // namespace - -void freeze_aten_len(std::shared_ptr graph) { - LOG(INFO) << "Running poros freeze_aten_len passes"; - FreezeAtenLen fal(std::move(graph)); - fal.run(); -} - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/lowering/try_to_freeze_aten_size.cpp b/poros/poros/lowering/try_to_freeze_aten_size.cpp deleted file mode 100644 index 344f609f20b..00000000000 --- a/poros/poros/lowering/try_to_freeze_aten_size.cpp +++ /dev/null @@ -1,226 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file try_to_freeze_aten_size.cpp -* @author tianjinjin@baidu.com -* @date Fri Nov 26 11:35:16 CST 2021 -* @brief -**/ - -#include "poros/lowering/lowering_pass.h" - -#include -#include -#include -#include -#include - -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -namespace { -using namespace torch::jit; - -bool has_type_and_dim(const Value* value) { - auto op = value->type()->cast(); - return op->sizes().size().has_value() && op->scalarType().has_value(); -} - -static std::string output_vec_size(const std::vector& vec_size) { - if (vec_size.empty()) { - return std::string(""); - } else { - std::string output_str = "["; - for (int64_t i : vec_size) { - output_str += (std::to_string(i) + std::string(", ")); - } - output_str.pop_back(); - output_str.pop_back(); - output_str.push_back(']'); - return output_str; - } -} - -struct FreezeAtenSize { - FreezeAtenSize(std::shared_ptr graph) : graph_(std::move(graph)) {} - - void run() { - GRAPH_DUMP("before freeze_aten_sizes Graph: ", graph_); - bool changed = freeze_aten_sizes(graph_->block()); - if (changed) { - ConstantPropagation(graph_); - EliminateDeadCode(graph_); - EliminateCommonSubexpression(graph_); - ConstantPooling(graph_); - } - GRAPH_DUMP("after freeze_aten_sizes Graph: ", graph_); - } - -private: - - bool is_aten_size_node(Node* node) { - if (node->kind() != aten::size) { - return false; - } - //TODO: may be add more check situation - return true; - } - - void replace_int_list(Node* node, const std::vector& inplace_number) { - LOG(INFO) << "try to replace the output of node :" << node_info(node) - << " with constant value " << output_vec_size(inplace_number); - torch::jit::WithInsertPoint guard(graph_->block()->nodes().front()); - auto int_list_const = graph_->insertConstant(inplace_number); - node->outputs().at(0)->replaceAllUsesWith(int_list_const); - } - - /** - * try to calculate the result of aten::slice. - * the schema of aten::slice is: - * "aten::slice(t[] l, int? start=None, int? end=None, int step=1) -> t[]" - * **/ - bool calculate_aten_slice(const torch::jit::Node* slice_node, - const std::vector& input, - std::vector& output) { - - if (slice_node->inputs().at(1)->node()->kind() != prim::Constant || - slice_node->inputs().at(2)->node()->kind() != prim::Constant || - slice_node->inputs().at(3)->node()->kind() != prim::Constant) { - return false; - } - - const int64_t input_len = input.size(); - auto maybe_start = toIValue(slice_node->inputs().at(1)); - auto start_index = maybe_start->isNone() ? 0 : maybe_start.value().toInt(); - const int64_t normalized_start = (start_index < 0) ? (input_len + start_index) : start_index; - - auto maybe_end = toIValue(slice_node->inputs().at(2)); - auto temp_end_index = maybe_end->isNone() ? INT64_MAX : maybe_end.value().toInt(); - auto end_idx = std::min(temp_end_index, input_len); - const int64_t normalized_end = (end_idx < 0) ? (input_len + end_idx) : end_idx; - - if (normalized_end <= normalized_start) { - return false; - } - int64_t step = toIValue(slice_node->inputs().at(3)).value().toInt(); - - output.reserve(normalized_end - normalized_start); - for (auto i = normalized_start; i < normalized_end;) { - output.push_back(input[i]); - i += step; - } - - LOG(INFO) << "calculate_aten_slice done, input size: " << output_vec_size(input) - << ", start_index: " << normalized_start - << ", end_index: " << normalized_end - << ", step: " << step - << ", ouput size: " << output_vec_size(output); - - auto it = std::find_if(output.begin(), output.end(), [&](const int64_t& v) {return v == -1;}); - //不满足条件,output中有-1的值,说明存在dynamic的dim,不能替换成常量。 - if (it != output.end()) { - return false; - } - - return true; - } - - /** - * @brief 尝试解析aten::size的数据 - * 如果aten::size返回的list的后续使用,可以解除与动态变化的维度的关系,则相应值进行常量替换。 - **/ - bool try_to_freeze_aten_size(Node* node) { - - std::vector dims; - if ((node->inputs()[0])->type()->isSubtypeOf(c10::TensorType::get()) && - has_type_and_dim(node->inputs()[0])) { - gen_dims_for_tensor(node->inputs()[0], dims); - } else { - return false; - } - if (node->inputs().size() == 2) { - return false; - } - - //输入非tensor的场景,根据输入类型节点的类型简单判断。 - auto output_value = node->outputs()[0]; // should be a int[] - auto users_count = (node->outputs()[0])->uses().size(); - - //situation one: 如果aten::size算子本身的计算结果里面没有-1,则表示该tensor非dynamic, - //则可以不管后面跟的是什么op,直接替换size的输出。 - auto it = std::find_if(dims.begin(), dims.end(), [&](const int64_t& v) {return v == -1;}); - if (it == dims.end()) { - LOG(INFO) << "aten size output memebers are all constant situation, dim info: " << output_vec_size(dims); - replace_int_list(node, dims); - node->destroy(); - return true; - } - - //situation two: aten::size is dynamic but the user is aten::slice - if (users_count == 1 && (output_value->uses()[0]).user->kind() == aten::slice) { - LOG(INFO) << "aten size user is aten::slice situation."; - auto slice_node = (output_value->uses()[0]).user; - std::vector sliced_list; - if (calculate_aten_slice(slice_node, dims, sliced_list)) { - //满足条件,替换节点 - replace_int_list(slice_node, sliced_list); - //slice node 可以析构掉了 - slice_node->destroy(); - //当前的aten::size节点也可以析构掉了 - node->destroy(); - return true; - } - } else { - LOG(INFO) << "not supported situation now."; - } - return false; - } - - bool freeze_aten_sizes(Block* block) { - bool changed = false; - //fix bug: 可能连续后面几个节点被删除(比如aten::slice + aten::size), iterator改成从后往前迭代。 - for (auto it = block->nodes().rbegin(); it != block->nodes().rend();) { - // we might destroy the current node, so we need to pre-increment - // the iterator - Node* node = *it; - ++it; - for (Block* subblock : node->blocks()) { - changed |= freeze_aten_sizes(subblock); - } - if (is_aten_size_node(node)) { - LOG(INFO) << "find aten::size node: " << node_info(node); - changed |= try_to_freeze_aten_size(node); - } - } - return changed; - } - - std::shared_ptr graph_; -}; - -} // namespace - -void freeze_aten_size(std::shared_ptr graph) { - LOG(INFO) << "Running poros freeze_aten_size passes"; - FreezeAtenSize fas(std::move(graph)); - fas.run(); -} - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/lowering/try_to_freeze_list_construct.cpp b/poros/poros/lowering/try_to_freeze_list_construct.cpp deleted file mode 100644 index bd6d3a2ecfa..00000000000 --- a/poros/poros/lowering/try_to_freeze_list_construct.cpp +++ /dev/null @@ -1,174 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file: try_to_freeze_array.cpp -* @author: zhangfan51@baidu.com -* @data: 2022-03-23 15:53:29 -* @brief: -**/ -#include "poros/lowering/lowering_pass.h" - -#include - -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -namespace { -using namespace torch::jit; -/** - * @brief 尝试展开for循环内list.append()的情况 - * try to expand - * graph(%feat.29 : Tensor): - * %256 : int = prim::Constant[value=2]() - * %257 : float = prim::Constant[value=2.]() - * %scale_factors.88 : float[] = prim::ListConstruct() - * = prim::Loop(%256, %2146) - * block0(%746 : int): - * %747 : float[] = aten::append(%scale_factors.88, %257) - * -> (%2146) - * as - * graph(%feat.29 : Tensor): - * %256 : int = prim::Constant[value=2]() - * %257 : float = prim::Constant[value=2.]() - * %scale_factors.88 : float[] = prim::ListConstruct() - * %747 : float[] = aten::append(%scale_factors.88, %257) - * %748 : float[] = aten::append(%scale_factors.88, %257) - **/ - - -struct FreeezeListConstruct { - FreeezeListConstruct(std::shared_ptr graph) : graph_(std::move(graph)) {} - - void run() { - try_to_freeze_list_construct(graph_->block()); - // 运行一遍常量折叠 - torch::jit::ConstantPropagation(graph_); - } - -private: - template - void replace_contant_list_construct(Node* node, std::vector &data_array) { - LOG(INFO) << "try to replace the output of node :" << node_info(node) - << " with constant value " << data_array; - torch::jit::WithInsertPoint guard(graph_->block()->nodes().front()); - auto list_const = graph_->insertConstant(data_array); - node->outputs().at(0)->replaceAllUsesWith(list_const); - } - - void try_to_freeze_list_construct(Block* block) { - auto it = block->nodes().begin(); - while (it != block->nodes().end()) { - auto node = *it; - ++it; //先++it, node 可能destroy掉。 - for (auto block : node->blocks()) { - try_to_freeze_list_construct(block); - } - - // 找到 prim::ListConstruct - if (node->kind() != prim::ListConstruct) { - continue; - } - // 判断 ListConstruct的output为float[] or int[] - if (!(node->outputs()[0])->type()->isSubtypeOf(c10::ListType::ofFloats()) && - !(node->outputs()[0])->type()->isSubtypeOf(c10::ListType::ofInts())) { - continue; - } - // 判断 ListConstruct的inputs应为空,否则会有值不相同 - if (node->inputs().size() != 0) { - continue; - } - //only do for float[] and int[] - // 判断该ListConstruct的所有使用者,是否仅做过一次aten::append修改 - int use_flag = 0; - Node* app_node = nullptr; - for (auto &use : node->outputs()[0]->uses()) { - if (use.user->owningBlock() != node->owningBlock() && - use.user->kind() == aten::append) { - use_flag++; - app_node = use.user; - } - } - if (use_flag != 1) { - continue; - } - // 判断append的block 是放在prim::Loop中的, 且在该Loop里只有1个block - Block* app_block = app_node->owningBlock(); - Node* loop_node = app_block->owningNode(); - // 目前先仅考虑owingNode为prim::Loop的情况,如后面遇到其他类似pattern再做相应判断调整 - if (loop_node->kind() != prim::Loop || loop_node->blocks().size() > 1) { - continue; - } - - auto app_it = app_block->nodes().begin(); - std::vector app_block_nodes; - while (app_it != app_block->nodes().end()) { - app_block_nodes.push_back(*app_it); - ++app_it; - } - // 仅处理形如这种的情况: - // block0(%746 : int): - // %747 : float[] = aten::append(%scale_factors.88, %257) - // -> (%2146) - // block 中仅包括1个append - if (app_block_nodes.size() != 1) { - LOG(INFO) << "freeze_list_construct: append block nodes size is more than 1. "; - continue; - } - // prim::Loop的 循环次数必须为prim::Constant, append的value也必须为prim::Constant. - if ((loop_node->inputs()[0])->node()->kind() != prim::Constant || - (app_node->inputs()[1])->node()->kind() != prim::Constant ) { - LOG(INFO) << "freeze_list_construct: append's input or loop's input type is not prim::Constant."; - continue; - } - auto loop_max = toIValue(loop_node->inputs()[0]->node()->output()).value().toInt(); - auto loop_cond = toIValue(loop_node->inputs()[1]->node()->output()).value().toBool(); - // loop_cond must be true here, check again. - if (!loop_cond) { - continue; - } - auto value = toIValue((app_node->inputs()[1])->node()->output()).value(); - if (value.isInt()) { - std::vector array_value(loop_max, value.toInt()); - replace_contant_list_construct(node, array_value); - } else if (value.isDouble()) { - std::vector array_value(loop_max, value.toDouble()); - replace_contant_list_construct(node, array_value); - } else { - continue; - } - //destroy app_block下所有node - for (size_t i = 0; i < app_block_nodes.size(); ++i) { - app_block_nodes[i]->destroy(); - } - // loop_node->destroy(); - } - } - std::shared_ptr graph_; -}; - -} // namespace - -void freeze_list_construct(std::shared_ptr graph) { - LOG(INFO) << "Running poros freeze_list_construct passes"; - FreeezeListConstruct flc(std::move(graph)); - flc.run(); -} - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/lowering/try_to_freeze_percentformat.cpp b/poros/poros/lowering/try_to_freeze_percentformat.cpp deleted file mode 100644 index 91985ded02b..00000000000 --- a/poros/poros/lowering/try_to_freeze_percentformat.cpp +++ /dev/null @@ -1,249 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file try_to_freeze_percentformat.cpp -* @author tianjinjin@baidu.com -* @date Wed Nov 24 15:13:00 CST 2021 -* @brief -**/ - -#include "poros/lowering/lowering_pass.h" - -#include -#include -#include -#include -#include -#include - -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -namespace { - -using namespace torch::jit; - -struct FreezePercentFormat { - FreezePercentFormat(std::shared_ptr graph) : graph_(std::move(graph)) {} - - void run() { - bool changed = freeze_percentformats(graph_->block()); - if (changed) { - ConstantPropagation(graph_); - EliminateDeadCode(graph_); - EliminateCommonSubexpression(graph_); - ConstantPooling(graph_); - //PeepholeOptimize(graph_, /*addmm_fusion_enabled*/false); - //CheckInplace(graph_); - //runRequiredPasses(graph_); - } - return; - } - -private: - - bool is_percent_format_node(Node* node) { - if (node->kind() != aten::percentFormat) { - return false; - } - //maybe to add more - return true; - } - - // IValue tags are intentionally private, so we need additional logic to cast - // the IValue type to the specified format. - void add_formatted_arg(char key, - const IValue& ival, - std::stringstream& ss, - int precision = 6) { - // TODO: Implement precison-based formatting - std::stringstream tmp; - switch (key) { - case 'd': - case 'i': - if (ival.isInt()) { - ss << ival.toInt(); - } else { - ss << static_cast(ival.toDouble()); - } - break; - case 'e': - case 'E': - tmp << std::setprecision(precision) << std::scientific; - if (key == 'E') { - tmp << std::uppercase; - } - if (ival.isInt()) { - tmp << static_cast(ival.toInt()); - } else { - tmp << static_cast(ival.toDouble()); - } - ss << tmp.str(); - break; - case 'f': - case 'F': - tmp << std::setprecision(precision) << std::fixed; - if (ival.isInt()) { - tmp << static_cast(ival.toInt()); - } else { - tmp << static_cast(ival.toDouble()); - } - ss << tmp.str(); - break; - case 'c': - if (ival.isInt()) { - ss << static_cast(ival.toInt()); - } else { - ss << ival.toStringRef(); - } - break; - case 's': - if (ival.isString()) { - ss << ival.toStringRef(); - } else { - ss << ival; - } - break; - default: - TORCH_CHECK(false, "The specifier %", key, " is not supported in TorchScript format strings"); - } - } - - std::string interprete_percent_format(std::vector& stack, size_t num_inputs) { - auto format_str = peek(stack, 0, num_inputs).toStringRef(); - auto args = last(stack, num_inputs - 1)[0]; - size_t args_size = 1; // assumed size - if (args.isTuple()) { - args_size = args.toTuple()->elements().size(); - } - std::stringstream ss; - size_t used_args = 0; - size_t begin = 0; - - while (true) { - size_t percent_idx = format_str.find('%', begin); - if (percent_idx == std::string::npos) { - ss << format_str.substr(begin); - break; - } - size_t format_idx = percent_idx + 1; - TORCH_CHECK( - percent_idx < format_str.length() - 1, "Incomplete format specifier"); - ss << format_str.substr(begin, percent_idx - begin); - - if (format_str.at(format_idx) == '%') { - ss << '%'; - begin = percent_idx + 2; // skip the `%` and the format specifier - continue; - } - - // NOLINTNEXTLINE(clang-diagnostic-sign-compare) - TORCH_CHECK(used_args < args_size, "Too few arguments for format string"); - char key = format_str.at(format_idx); - IValue arg; - if (args.isTuple()) { - arg = args.toTuple()->elements()[used_args]; - } else { - arg = args; - } - add_formatted_arg(key, arg, ss); - begin = percent_idx + 2; - ++used_args; - } - // NOLINTNEXTLINE(clang-diagnostic-sign-compare) - TORCH_CHECK(used_args == args_size, "Too many arguments for format string"); - std::string result = ss.str(); - return result; - } - - /** - * the schema of percentformat is : "aten::percentFormat(str self, ...) -> str" - * **/ - bool try_to_freeze_percentformat(Node* format_node) { - //Graph* graph = format_node->owningGraph(); - at::ArrayRef inputs = format_node->inputs(); - size_t num_inputs = inputs.size(); - - //no format input situation. - if (num_inputs < 2) { - LOG(INFO) << "should not freeze node: " << node_info(format_node); - return false; - } - - //bool all_input_constant = true; - std::vector stack; - for(size_t index = 0; index < num_inputs; index++) { - if (inputs[index]->node()->kind() != prim::Constant) { - LOG(INFO) << "should not freeze node: " << node_info(format_node); - return false; - } else { - c10::optional ivalue = toIValue(inputs[index]->node()->output()); - if (ivalue.has_value()) { - stack.push_back(ivalue.value()); - } - } - } - - if (stack.size() != num_inputs) { - LOG(INFO) << "should not freeze node: " << node_info(format_node); - return false; - } - - //if we reach here, that means all inputs are constant. let's calculate the result - std::string result = interprete_percent_format(stack, num_inputs); - LOG(INFO) << "try to replace the output of node :" << node_info(format_node) - << " with constant value: " << result; - WithInsertPoint guard(graph_->block()->nodes().front()); - Value* string_const = graph_->insertConstant(result); - format_node->outputs().at(0)->replaceAllUsesWith(string_const); - format_node->destroy(); - return true; - } - - bool freeze_percentformats(Block* block) { - bool changed = false; - for (auto it = block->nodes().begin(); it != block->nodes().end();) { - // we might destroy the current node, so we need to pre-increment - // the iterator - Node* node = *it; - ++it; - for (Block* subblock : node->blocks()) { - changed |= freeze_percentformats(subblock); - } - if (is_percent_format_node(node)) { - LOG(INFO) << "meet percent format node :" << node_info(node); - changed |= try_to_freeze_percentformat(node); - } - } - return changed; - } - -std::shared_ptr graph_; -}; - -} // namespace - -void freeze_percentformat(std::shared_ptr graph) { - LOG(INFO) << "Running poros freeze_percentformat passes"; - FreezePercentFormat fpf(std::move(graph)); - fpf.run(); -} - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/lowering/unpack_certain_ops.cpp b/poros/poros/lowering/unpack_certain_ops.cpp deleted file mode 100644 index 7c0c82da4f7..00000000000 --- a/poros/poros/lowering/unpack_certain_ops.cpp +++ /dev/null @@ -1,127 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -// Part of the following code in this file refs to -// https://github.com/pytorch/TensorRT/blob/master/core/lowering/passes/unpack_var.cpp -// -// Copyright (c) 2020-present, NVIDIA CORPORATION. All rights reserved. -// Copyright (c) Meta Platforms, Inc. and affiliates. -// Licensed under the BSD 3-Clause "New" or "Revised" License - -/** -* @file unpack_certain_ops.cpp -* @author tianjinjin@baidu.com -* @date Thu Sep 23 20:15:53 CST 2021 -* @brief -**/ - -#include "poros/lowering/lowering_pass.h" - -#include "torch/csrc/jit/passes/subgraph_rewrite.h" - -namespace baidu { -namespace mirana { -namespace poros { - -void unpack_std(std::shared_ptr& graph) { - std::string std_pattern = R"IR( - graph(%1, %dim, %unbiased, %keepdim): - %out: Tensor = aten::std(%1, %dim, %unbiased, %keepdim) - return (%out))IR"; - - std::string unpacked_pattern = R"IR( - graph(%1, %dim, %unbiased, %keepdim): - %z: Tensor = aten::var(%1, %dim, %unbiased, %keepdim) - %out: Tensor = aten::sqrt(%z) - return (%out))IR"; - - torch::jit::SubgraphRewriter std_rewriter; - std_rewriter.RegisterRewritePattern(std_pattern, unpacked_pattern); - std_rewriter.runOnGraph(graph); -} - -void unpack_var(std::shared_ptr& graph) { - std::string var_pattern = R"IR( - graph(%input, %dim, %unbiased, %keepdim): - %out: Tensor = aten::var(%input, %dim, %unbiased, %keepdim) - return (%out))IR"; - std::string unpacked_pattern = R"IR( - graph(%input, %dims, %unbiased, %keepdim): - %none: None = prim::Constant() - %false: bool = prim::Constant[value=0]() - %0: int = prim::Constant[value=0]() - %f32_dtype: int = prim::Constant[value=6]() - %1: int = prim::Constant[value=1]() - %sqrd: Tensor = aten::mul(%input, %input) - %sqrdmean: Tensor = aten::mean(%sqrd, %dims, %keepdim, %none) - %mean: Tensor = aten::mean(%input, %dims, %keepdim, %none) - %meansqrd: Tensor = aten::mul(%mean, %mean) - %var: Tensor = aten::sub(%sqrdmean, %meansqrd, %1) - %varout : Tensor = prim::If(%unbiased) - block0(): - %shape: int[] = aten::size(%input) - %shapet: Tensor = aten::tensor(%shape, %f32_dtype, %none, %false) - %dim: int = prim::ListUnpack(%dims) - %reduceddims: Tensor = aten::select(%shapet, %0, %dim) - %numel: Tensor = aten::prod(%reduceddims, %dim, %keepdim, %none) - %mul: Tensor = aten::mul(%var, %numel) - %sub: Tensor = aten::sub(%numel, %1, %1) - %v: Tensor = aten::div(%mul, %sub) - -> (%v) - block1(): - -> (%var) - return(%varout))IR"; - - torch::jit::SubgraphRewriter var_rewriter; - var_rewriter.RegisterRewritePattern(var_pattern, unpacked_pattern); - var_rewriter.runOnGraph(graph); -} - -void replace_log_softmax(std::shared_ptr graph) { - std::string old_pattern = R"IR( - graph(%1, %dim, %dtype): - %out: Tensor = aten::log_softmax(%1, %dim, %dtype) - return (%out))IR"; - - std::string new_pattern = R"IR( - graph(%1, %dim, %dtype): - %2: Tensor = aten::softmax(%1, %dim, %dtype) - %out: Tensor = aten::log(%2) - return (%out))IR"; - - torch::jit::SubgraphRewriter std_rewriter; - std_rewriter.RegisterRewritePattern(old_pattern, new_pattern); - std_rewriter.runOnGraph(graph); -} - -void replace_log_sigmoid(std::shared_ptr graph) { - std::string old_pattern = R"IR( - graph(%1): - %out: Tensor = aten::log_sigmoid(%1) - return (%out))IR"; - - std::string new_pattern = R"IR( - graph(%1): - %2: Tensor = aten::sigmoid(%1) - %out: Tensor = aten::log(%2) - return (%out))IR"; - - torch::jit::SubgraphRewriter std_rewriter; - std_rewriter.RegisterRewritePattern(old_pattern, new_pattern); - std_rewriter.runOnGraph(graph); -} - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/lowering/unrolling_loop.cpp b/poros/poros/lowering/unrolling_loop.cpp deleted file mode 100644 index 29cf66d29db..00000000000 --- a/poros/poros/lowering/unrolling_loop.cpp +++ /dev/null @@ -1,386 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -// Part of the following code in this file refs to -// https://github.com/pytorch/pytorch/blob/master/torch/csrc/jit/passes/loop_unrolling.cpp -// -// Copyright (c) Meta Platforms, Inc. and affiliates. -// Licensed under the 3-Clause BSD License - -/** -* @file unrolling_loop.cpp -* @author tianjinjin@baidu.com -* @date Mon Nov 22 16:59:25 CST 2021 -* @brief this file is modified from torch/csrc/jit/passes/loop_unrolling.cpp -* and some parameters are different from the original funciton -**/ -#include "poros/lowering/lowering_pass.h" - -#include -#include -#include -#include -#include - -#include "poros/util/poros_util.h" - -namespace baidu { -namespace mirana { -namespace poros { - -namespace { -using namespace torch::jit; - -static constexpr int64_t UnrollFactor = 8; -static constexpr int64_t MaxBodySize = 256; -static constexpr int64_t MaxBodyRepeats = 64; -static constexpr int64_t MaxLoopMulResult = 32 * 64; - -struct UnrollingLoop { - UnrollingLoop(std::shared_ptr graph) : graph_(std::move(graph)) {} - - void run() { - bool changed = unroll_loops(graph_->block(), true); - GRAPH_DUMP("afte unroll_loop graph:", graph_); - changed |= eliminate_useless_loop_count_body(graph_->block()); - GRAPH_DUMP("afte eliminate_useless_loop_count_body graph:", graph_); - if (changed) { - ConstantPropagation(graph_); - EliminateDeadCode(graph_); - EliminateCommonSubexpression(graph_); - ConstantPooling(graph_); - } - return; - } - -private: - - bool is_for_loop(Node* node) { - if (node->kind() != prim::Loop) { - return false; - } - Value* start_cond = node->inputs().at(1); - c10::optional maybe_start_value = constant_as(start_cond); - Value* continue_cond = node->blocks().at(0)->outputs().at(0); - c10::optional maybe_continue_value = constant_as(continue_cond); - return maybe_start_value && *maybe_start_value && maybe_continue_value && *maybe_continue_value; - } - - int64_t limited_block_size(Block* body, int64_t limit) { - auto it = body->nodes().begin(); - auto end = body->nodes().end(); - for (int64_t i = 0; i < limit; ++it) { - for (Block* subblock : it->blocks()) { - i += limited_block_size(subblock, limit - i); - } - if (!it->notExecutedOp()) { - ++i; - } - if (it == end) { - return i; - } - } - return limit; - } - - int64_t calculate_block_size(Block* body) { - auto it = body->nodes().begin(); - int64_t count = 0; - while (it != body->nodes().end()) { - auto node = *it; - ++it; //先++it - for (auto block : node->blocks()) { - count += calculate_block_size(block); - } - if (!node->notExecutedOp()) { - ++count; - } - } - return count; - } - - bool is_small_block(Block* body) { - return limited_block_size(body, MaxBodySize + 1) <= MaxBodySize; - } - - // XXX: This function can only be called with a loop that is guaranteed to - // execute EXACTLY ONCE. - void inline_body(Node* loop) { - auto graph = loop->owningGraph(); - auto body = loop->blocks().at(0); - WithInsertPoint insert_point_guard{loop}; - - std::unordered_map value_map; - auto get_value = [&](Value* v) { - auto it = value_map.find(v); - if (it != value_map.end()) - return it->second; - return v; - }; - - // Loop node has extra (max_iters, initial_cond) inputs, - // body has an extra (loop_counter) input. - for (size_t i = 2; i < loop->inputs().size(); ++i) { - value_map[body->inputs()[i - 1]] = loop->inputs()[i]; - } - - for (Node* orig : body->nodes()) { - Node* clone = graph->insertNode(graph->createClone(orig, get_value)); - for (size_t i = 0; i < orig->outputs().size(); ++i) { - value_map[orig->outputs()[i]] = clone->outputs()[i]; - } - } - for (size_t i = 0; i < loop->outputs().size(); ++i) { - loop->outputs().at(i)->replaceAllUsesWith( - get_value(body->outputs().at(i + 1))); - } - // XXX: it is extremely important to destroy the loop in here. DCE might not - // be able to conclude that it's safe, because the loop might contain side - // effects. - loop->destroy(); - } - - // inserts a copy of body, passing inputs to the inputs of the block - // it returns the a list of the Values for the output of the block - std::vector insert_block_copy(Graph& graph, - Block* body, - at::ArrayRef inputs) { - TORCH_INTERNAL_ASSERT(inputs.size() == body->inputs().size()); - std::unordered_map value_map; - auto get_value = [&](Value* v) { - auto it = value_map.find(v); - if (it != value_map.end()) - return it->second; - return v; - }; - auto inputs_it = inputs.begin(); - for (Value* input : body->inputs()) { - value_map[input] = *inputs_it++; - } - for (Node* node : body->nodes()) { - Node* new_node = graph.insertNode(graph.createClone(node, get_value)); - auto outputs_it = new_node->outputs().begin(); - for (Value* output : node->outputs()) { - value_map[output] = *outputs_it++; - } - } - return fmap(body->outputs(), get_value); //maybe not recognized - } - - void repeat_body(Block* body, size_t times, Block* dest) { - auto graph = body->owningGraph(); - WithInsertPoint insert_point_guard(dest); - for (Value* input : body->inputs()) { - dest->addInput()->copyMetadata(input); - } - - std::vector io = dest->inputs().vec(); - TORCH_INTERNAL_ASSERT( - !body->inputs().at(0)->hasUses(), "loop counter should be unused"); - for (size_t i = 0; i < times; ++i) { - io[0] = body->inputs().at(0); - io = insert_block_copy(*graph, body, io); - } - for (Value* output : io) { - dest->registerOutput(output); - } - - // It's likely that we have some dead nodes now - for example the "true" - // constant that prevents the loop from breaking. We shouldn't wait too long - // before removing them because they might artificially increase the loop size - // and prevent outer loop unrolling. - torch::jit::EliminateDeadCode(dest, false); - } - - // Replaces the builtin loop counter with a "mutable" variable outside of the - // loop. - void replace_loop_counter(Node* loop) { - Graph* graph = loop->owningGraph(); - Block* body = loop->blocks().at(0); - WithInsertPoint guard(loop); - Value* init_counter = graph->insertConstant(0); - - loop->insertInput(2, init_counter); - loop->insertOutput(0)->setType(IntType::get()); - - Value* internal_counter = body->insertInput(1)->setType(init_counter->type()); - body->inputs()[0]->replaceAllUsesWith(internal_counter); - - WithInsertPoint insertPointGuard{body->return_node()}; - Value* result = graph->insert(aten::add, {internal_counter, 1}); - body->insertOutput(1, result); - } - - bool unroll(Node* loop) { - Graph* graph = loop->owningGraph(); - Block* body = loop->blocks().at(0); - - int64_t block_size = calculate_block_size(body); - if (block_size > MaxBodySize) { - return false; - } - - // if (!is_small_block(body)) { - // return false; - // } - - // We will be using a "mutable" counter outside of the loop instead of the - // default one, because this will allow us to share it between the unrolled - // loop and its epilogue. This is necessary only if the loop counter is - // actually used in the body. - if (body->inputs()[0]->uses().size() > 0) - replace_loop_counter(loop); - - // Some optimization for constant-length loops. If we know they won't run too - // many times, then we can unroll them entirely. - Value* trip_count = loop->inputs().at(0); - c10::optional const_len = constant_as(trip_count); - //auto loop_mul_result = block_size * const_len; - if (const_len && *const_len < MaxBodyRepeats && (block_size * (*const_len)) < MaxLoopMulResult) { - Block* dest = loop->addBlock(); - repeat_body(body, *const_len, dest); - loop->eraseBlock(0); - inline_body(loop); - return true; - } - - WithInsertPoint insert_point_guard{loop}; - - // Clone the loop before we unroll it. The clone will become the epilogue. - Node* loop_epilogue = - graph->createClone(loop, [](Value* v) { return v; })->insertAfter(loop); - for (size_t i = 0; i < loop->outputs().size(); ++i) { - loop->outputs()[i]->replaceAllUsesWith(loop_epilogue->outputs()[i]); - loop_epilogue->replaceInput(i + 2, loop->outputs()[i]); - } - - Block* dest = loop->addBlock(); - repeat_body(body, UnrollFactor, dest); - loop->eraseBlock(0); - - // Change the iteration counts of both loops - Value* iter_count = loop->inputs().at(0); - Value* unrolled_iter_count = graph->insert( - aten::__round_to_zero_floordiv, {iter_count, UnrollFactor}); - loop->replaceInput(0, unrolled_iter_count); - loop_epilogue->replaceInput( - 0, - graph->insert( - aten::sub, - {iter_count, - graph->insert(aten::mul, {unrolled_iter_count, UnrollFactor})})); - return true; - } - - bool unroll_loops(Block* block, bool constant_only) { - bool changed = false; - for (auto it = block->nodes().begin(); it != block->nodes().end();) { - // XXX: unroll might destroy the current node, so we need to pre-increment - // the iterator - Node* node = *it; - ++it; - for (Block* subblock : node->blocks()) { - changed |= unroll_loops(subblock, constant_only); - } - if (!is_for_loop(node)) { - continue; - } - //only handle max loop is constant situation. - if (constant_only && node->inputs().at(0)->node()->kind() != prim::Constant) { - continue; - } - changed |= unroll(node); - } - return changed; - } - - // 去掉像以下形式的prim::loop计数block - // %943 : int = prim::Loop(%65, %4641, %idx.5) - // block0(%944 : int, %945 : int): - // %0 : int = prim::Constant[value=1]() - // %7285 : int = aten::add(%945, %0) - // %947 : bool = aten::lt(%7285, %idx.13) - // %948 : bool = aten::__and__(%947, %4641) - // -> (%948, %7285) - // 其中原来的节点已经展开到parent graph中了,只剩下计数模块,没有实质作用,可以删掉。 - bool eliminate_useless_loop_count_body(Block* block) { - bool changed = false; - for (auto it = block->nodes().begin(); it != block->nodes().end();) { - // XXX: unroll might destroy the current node, so we need to pre-increment - // the iterator - Node* node = *it; - ++it; - for (Block* subblock : node->blocks()) { - changed |= eliminate_useless_loop_count_body(subblock); - } - // pattern - if (!is_uesless_loop_count_body(node)) { - continue; - } - changed |= destory_useles_loop_count_body(node); - } - return changed; - } - // 判断是否是无用的prim::loop计数模块 - bool is_uesless_loop_count_body(Node* node) { - // 输入node类型必须是prim::loop - if (node->kind() != prim::Loop) { - return false; - } - // prim::loop的输出必须无user - if (node->hasUses()) { - return false; - } - auto loop_block = node->blocks().at(0); - std::vector loop_block_nodes; - for (auto it = loop_block->nodes().begin(); it != loop_block->nodes().end(); it++) { - loop_block_nodes.push_back(*it); - if (loop_block_nodes.size() > 3) { - return false; - } - } - // block必须只有3个nodes且顺序必须是1-->aten::add、2-->aten::lt、3-->aten::__and__ - if (loop_block_nodes.size() == 3 && - loop_block_nodes[0]->kind() == aten::add && - loop_block_nodes[1]->kind() == aten::lt && - loop_block_nodes[2]->kind() == aten::__and__) { - LOG(INFO) << "Find useless loop counter body on node: [ " << node_info(node) << " ]"; - return true; - } - return false; - } - // 删掉无用的prim::loop计数节点 - bool destory_useles_loop_count_body(Node* node) { - if (node->kind() != prim::Loop) { - return false; - } - LOG(INFO) << "Destory useless loop counter node: [ " << node_info(node) << " ]"; - node->destroy(); - return true; - } - - std::shared_ptr graph_; -}; - -} // namespace - -void unrolling_loop(std::shared_ptr graph) { - LOG(INFO) << "Running poros unrolling_loop passes"; - UnrollingLoop ul(std::move(graph)); - ul.run(); -} - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/util/graph_test_helper.cpp b/poros/poros/util/graph_test_helper.cpp deleted file mode 100644 index bf01b69d3ce..00000000000 --- a/poros/poros/util/graph_test_helper.cpp +++ /dev/null @@ -1,167 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file test_util.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include "graph_test_helper.h" - -#include -#include - -namespace baidu { -namespace mirana { -namespace poros { -namespace graphtester { - -static inline void clone_tensor_vector(const std::vector &old_vec, std::vector &new_vec) { - for (auto i: old_vec) { - new_vec.push_back(i.clone()); - } -} - - -static bool write_to_log(const std::string &log_path, const std::string &inform) { - if (log_path.empty()) { - return true; - } else { - std::ofstream log_file(log_path, std::ios::app); - if (log_file) { - log_file << inform; - log_file.close(); - return true; - } else { - return false; - } - } -}; - - -static std::vector run_graph(const std::shared_ptr &graph, - const std::vector &input_data, - const baidu::mirana::poros::PorosOptions &poros_option, - const std::string &log_path) { - std::vector graph_output; - // 构建graph exe - std::string function_name("test tensor"); - torch::jit::GraphExecutor graph_exe(graph, function_name); - // 将输入导入ivalue vector - std::vector graph_input; - for (size_t i = 0; i < input_data.size(); i++) { - graph_input.push_back(input_data[i]); - } - // 执行graph - std::clock_t start, end; - start = std::clock(); - graph_exe.run(graph_input); - end = std::clock(); - std::string log_inform = "graph time:" + std::to_string(double(end - start) / CLOCKS_PER_SEC * 1000.0) + "ms\t"; - std::cout << log_inform; - if (!write_to_log(log_path, log_inform)) { - LOG(WARNING) << "write to log failed"; - } - // 提取结果 - for (size_t i = 0; i < graph_input.size(); i++) { - auto tmp_ivalue = graph_input[i]; - graph_output.push_back(tmp_ivalue.toTensor()); - } - return graph_output; -}; - - -std::vector -replace_input_tensor_to_constant(std::shared_ptr graph, const std::vector &input_data, - const std::vector &input_data_type_mask) { - torch::jit::WithInsertPoint guard(graph->block()->nodes().front()); - std::vector graph_input_tensor; - std::vector eraseInputIdx; - for (size_t i = 0; i < input_data_type_mask.size() && i < graph->inputs().size() && i < input_data.size(); i++) { - switch (input_data_type_mask[i]) { - case InputTensor: //正常输入Tensor - graph_input_tensor.push_back(input_data[i].toTensor()); - break; - case ConstantTensor: //op的weights和bias - graph->inputs()[i]->replaceAllUsesWith(graph->insertConstant(input_data[i])); - eraseInputIdx.push_back(i); - break; - case ConstantIntVector: // int[] = prim::Constant[value=[1, 1, 1]]() - graph->inputs()[i]->replaceAllUsesWith(graph->insertConstant(input_data[i].toIntList())); - eraseInputIdx.push_back(i); - break; - } - } - // 从后向前删除多余的input - for (auto it = eraseInputIdx.rbegin(); it != eraseInputIdx.rend(); it++) { - graph->eraseInput(*it); - } - - return graph_input_tensor; -} - -bool run_graph_and_fused_graph(const std::string &graph_IR, - const baidu::mirana::poros::PorosOptions &poros_option, - std::shared_ptr fuser, - const std::vector &input_data, - const std::vector &input_data_type_mask, - std::vector &original_graph_output, - std::vector &fused_graph_output, - std::string log_path) { - try { - fuser->reset(); - // 解析graph - auto graph = std::make_shared(); - torch::jit::parseIR(graph_IR, graph.get()); - auto input_tensor = replace_input_tensor_to_constant(graph, input_data, input_data_type_mask); - // 冷启动运行原始graph - std::vector input_of_replaced_graph; - clone_tensor_vector(input_tensor, input_of_replaced_graph); - std::cout << "input replaced "; - original_graph_output = run_graph(graph, input_of_replaced_graph, poros_option, log_path); - - // 运行常量化后的graph - std::vector input_of_ori_graph; - clone_tensor_vector(input_tensor, input_of_ori_graph); - std::cout << std::endl << "original "; - original_graph_output = run_graph(graph, input_of_ori_graph, poros_option, log_path); - - // 运行fuse后的graph - std::vector input_of_fused_graph; - clone_tensor_vector(input_tensor, input_of_fused_graph); - fuser->fuse(graph); - std::cout << std::endl << "op fused "; - fused_graph_output = run_graph(graph, input_of_fused_graph, poros_option, log_path); - std::cout << std::endl << fuser->info() << std::endl << std::endl; - } catch (...) { - return false; - } - return true; -} - -bool almost_equal(const at::Tensor &a, const at::Tensor &b, const float &threshold) { - auto a_float = a.toType(at::kFloat); - auto b_float = b.toType(at::kFloat); - double maxValue = 0.0; - maxValue = fmax(a.abs().max().item(), maxValue); - maxValue = fmax(b.abs().max().item(), maxValue); - at::Tensor diff = a_float - b_float; - return diff.abs().max().item() <= threshold * maxValue; -} - -}// namespace graphtester -}// namespace poros -}// namespace mirana -}// namespace baidu diff --git a/poros/poros/util/graph_test_helper.h b/poros/poros/util/graph_test_helper.h deleted file mode 100644 index 625e08d9e72..00000000000 --- a/poros/poros/util/graph_test_helper.h +++ /dev/null @@ -1,73 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file test_util.h -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#pragma once - -#include "poros/compile/poros_module.h" -#include "poros/lowering/op_fuse_pass.h" - -namespace baidu { -namespace mirana { -namespace poros { -namespace graphtester { - - -enum InputTypeEnum { - InputTensor = 0, // op 的输入 - ConstantTensor, // op的权重和偏置 - ConstantIntVector, // 如conv2d的stride等要求int[]的参数 -}; - -/** - * - * @param graph_IR - * @param poros_option : default device is GPU - * @param fuser : op fuser to test - * @param input_data : vector of at::IValue, which Compatible with Tensor, vector , scalar , etc. - * @param input_data_type_mask : tell the func how to deal with the input data, used to trans Tensor, vector or []int to prim::Constant - * @param original_graph_output : vector of at::Tensor, graph outputs - * @param fused_graph_output : vector of at::Tensor, poros outputs - * @param log_path - * @return bool - */ -bool run_graph_and_fused_graph(const std::string &graph_IR, - const baidu::mirana::poros::PorosOptions &poros_option, - std::shared_ptr fuser, - const std::vector &input_data, - const std::vector &input_data_type_mask, - std::vector &original_graph_output, - std::vector &fused_graph_output, - std::string log_path = ""); - -/** - * @brief compare the similarity of two Tensors containing Float - * - * @param [in] a : first Tensor - * @param [in] b : second Tensor - * @param [in] threshold : acceptable relative threshold - * @return bool - * @retval true => succeed false => failed -**/ -bool almost_equal(const at::Tensor &a, const at::Tensor &b, const float &threshold); - -}// namespace graphtester -}// namespace poros -}// namespace mirana -}// namespace baidu diff --git a/poros/poros/util/macros.h b/poros/poros/util/macros.h deleted file mode 100644 index a6888bfbbc5..00000000000 --- a/poros/poros/util/macros.h +++ /dev/null @@ -1,77 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file macros.h -* @author tianjinjin@baidu.com -* @date Fri Jun 4 16:16:38 CST 2021 -* @brief -**/ - -#pragma once - -#include - -#include - -namespace baidu { -namespace mirana { -namespace poros { - -#define POROS_TIME_COST_US(pre, now) ((now.tv_sec - pre.tv_sec) * 1000000 + (now.tv_usec - pre.tv_usec)) -#define POROS_TIME_COST_MS(pre, now) ((now.tv_sec - pre.tv_sec) * 1000 + (now.tv_usec - pre.tv_usec) / 1000) - -// ---------------------------------------------------------------------------- -// Error reporting macros -// ---------------------------------------------------------------------------- -#define POROS_CHECK_RET_EXIT(n, s) { \ - if ((n) != 0) { \ - LOG(FATAL) << s; \ - exit(1); \ - } \ -} - -#define POROS_CHECK_RET(n, s) { \ - if ((n) != 0) { \ - LOG(WARNING) << s; \ - return -1; \ - } \ -} - -#define POROS_CHECK_TRUE(n, s) { \ - if ((n) != true) { \ - LOG(WARNING) << s; \ - return false; \ - } \ -} - -#define POROS_THROW_ERROR(msg) \ - throw ::c10::Error({__func__, __FILE__, static_cast(__LINE__)}, #msg); - -#define POROS_ASSERT(cond, ...) \ - if (!(cond)) { \ - POROS_THROW_ERROR( \ - #cond << " ASSERT FAILED at " << __FILE__ << ':' << __LINE__ \ - << ", consider filing a bug to cudp@baidu.com \n" \ - << __VA_ARGS__); \ - } - -#define POROS_CHECK(cond, ...) \ - if (!(cond)) { \ - POROS_THROW_ERROR("Expected " << #cond << " to be true but got false\n" << __VA_ARGS__); \ - } - -} // namespace poros -} // namespace mirana -} // namespace baidu \ No newline at end of file diff --git a/poros/poros/util/poros_util.cpp b/poros/poros/util/poros_util.cpp deleted file mode 100644 index 97e7a6dd648..00000000000 --- a/poros/poros/util/poros_util.cpp +++ /dev/null @@ -1,339 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file poros_util.cpp -* @author tianjinjin@baidu.com -* @date Wed Apr 7 17:52:36 CST 2021 -* @brief -**/ - -#include "poros/util/poros_util.h" -#include "poros/log/poros_logging.h" -#include "poros/util/macros.h" -#include "poros/context/poros_global.h" - -namespace baidu { -namespace mirana { -namespace poros { - -int merge_graph_to_module(std::shared_ptr& to_merge_graph, - torch::jit::Module& module, - bool init_module_ptr) { - if (init_module_ptr) { - //先把model本身作为graph的第一个参数传进去, 此处勿忘!!!!!! - auto self = to_merge_graph->insertInput(0, "self"); - self->setType(module._ivalue()->type()); - } - - auto new_method = module._ivalue()->compilation_unit()->create_function("forward", to_merge_graph); - std::vector args; - int index = 0; - for (auto in : to_merge_graph->inputs()) { - args.push_back(c10::Argument("input" + std::to_string(index), in->type())); - index++; - } - - index = 0; - std::vector res; - for (auto out : to_merge_graph->outputs()) { - res.push_back(c10::Argument("output" + std::to_string(index), out->type())); - index++; - } - auto schema = c10::FunctionSchema(new_method->name(), new_method->name(), args, res); - module.type()->addMethod(new_method); - new_method->setSchema(schema); - return 0; -} - -torch::jit::Module build_tmp_module(std::shared_ptr& sub_graph) { - torch::jit::script::Module new_mod("tmp_submodule"); - auto graph = sub_graph->copy(); - merge_graph_to_module(graph, new_mod, true); - return new_mod; -} - -//好好参考aten/src/ATen/core/type.cpp 里 operator<< 的写法 -bool gen_dims_for_tensor(const torch::jit::Value* value, std::vector& dims) { - POROS_CHECK_TRUE((value->type()->isSubtypeOf(c10::TensorType::get())), - "given value for gen_dims_for_tensor is not Tensor as expected"); - std::vector().swap(dims); - c10::TensorTypePtr op = value->type()->cast(); - if (auto ndim = op->sizes().size()) { - for (size_t i = 0; i < *ndim; ++i) { - if (auto s = op->sizes()[i]) { - dims.push_back(s.value()); - } else { - dims.push_back(-1); - } - } - return true; - } - return false; -} - -void update_global_context(torch::jit::Value* old_value, torch::jit::Value* new_value) { - // copy value_dynamic_shape_map - if (old_value->type()->isSubtypeOf(c10::TensorType::get())) { - if (PorosGlobalContext::instance()._value_dynamic_shape_map.count(old_value) > 0) { - PorosGlobalContext::instance()._value_dynamic_shape_map[new_value] = - PorosGlobalContext::instance()._value_dynamic_shape_map[old_value]; - } - } else if (old_value->type()->kind() == c10::TypeKind::IntType) { - update_global_int_intlist_map_context(old_value, new_value); - } else if (old_value->type()->kind() == c10::TypeKind::ListType) { - if (old_value->type()->isSubtypeOf(c10::ListType::ofInts())) { - update_global_int_intlist_map_context(old_value, new_value); - } - PorosGlobalContext::instance()._list_size_map.update_value(old_value, new_value); - } else { - - } - //to add more @wangrui39 -} - -void update_global_int_intlist_map_context(torch::jit::Value* old_value, torch::jit::Value* new_value) { - if (PorosGlobalContext::instance()._int_intlist_values_map.count(old_value) > 0) { - PorosGlobalContext::instance()._int_intlist_values_map[new_value] = - PorosGlobalContext::instance()._int_intlist_values_map[old_value]; - } -} - -void update_global_list_size_map_node_key_context(torch::jit::Node* old_node, torch::jit::Node* new_node) { - PorosGlobalContext::instance()._list_size_map.update_node(old_node, new_node); -} - -void unmerge_subgraph(torch::jit::Node* subgraph_node) { - // Inline the graph, replace uses of node outputs and destroy the node - auto outer_graph = subgraph_node->owningGraph(); - std::shared_ptr sub_graph = subgraph_node->g(torch::jit::attr::Subgraph); - - torch::jit::WithInsertPoint guard(subgraph_node); - const auto subgraph_outputs = torch::jit::insertGraph( - *outer_graph, *sub_graph, subgraph_node->inputs()); - AT_ASSERT(subgraph_outputs.size() >= subgraph_node->outputs().size()); - for (size_t i = 0; i < subgraph_node->outputs().size(); ++i) { - subgraph_node->outputs()[i]->replaceAllUsesWith(subgraph_outputs[i]); - update_global_context(subgraph_node->outputs()[i], subgraph_outputs[i]); - } - subgraph_node->destroy(); -} - -void find_to_optimized_nodes(torch::jit::Block* block, std::vector& to_optimized_nodes) { - //bool changed = false; - for (auto it = block->nodes().begin(); it != block->nodes().end(); it++) { - torch::jit::Node* node = *it; - for (torch::jit::Block* subblock : node->blocks()) { - find_to_optimized_nodes(subblock, to_optimized_nodes); - } - if (node->kind() == torch::jit::prim::CudaFusionGroup) { - to_optimized_nodes.push_back(node); - } - } -} - - -/******************************************************************** - SOME DEPRECATED FUNCTIONS BELOW -*********************************************************************/ -//DEPRECATED -bool gen_dims_for_scarlar(const torch::jit::Value* value, std::vector& dims) { - return false; -} - -// gen dims for tensorlist input -// DEPRECATED -bool gen_dims_for_tensorlist(const torch::jit::Value* value, std::vector& dims) { - // if we want to treat the tensorlist as a single input to tensort. - // TODO: we should check the tensors in list are of the same size. - // std::vector pre_dims; - POROS_CHECK_TRUE(value->type()->isSubtypeOf(c10::ListType::ofTensors()), - "given value for gen_dims_for_tensorlist is not TensorList as expected"); - std::vector().swap(dims); - - auto producer = value->node(); - if (producer->kind() == torch::jit::aten::meshgrid) { - LOG(INFO) << "to support: torch::jit::aten::meshgrid"; - - // prim::ListConstruct - } else if (producer->kind() == torch::jit::prim::ListConstruct) { - //situation one: some node like : - // %out : Tensor[] = prim::ListConstruct(%intput) - if (producer->inputs().size() > 0) { - auto op = producer->inputs()[0]->type()->cast(); - if (op->sizes().size().has_value() && op->scalarType().has_value()) { - dims = op->sizes().concrete_sizes().value(); - return true; - } - //situation two: some node like: - // %out : Tensor[] = prim::ListConstruct() - // %new_out: Tensor[] = aten::append(%out, %item) - } else { - for (auto use: value->uses()) { - LOG(INFO) << "checking user: " << node_info_with_attr(use.user); - if(use.user->kind() == torch::jit::aten::append && - use.user->inputs().size() > 1) { - auto op = use.user->inputs()[1]->type()->cast(); - if (op->sizes().size().has_value() && op->scalarType().has_value()) { - dims = op->sizes().concrete_sizes().value(); - return true; - } - } - } - } - LOG(INFO) << "to support: torch::jit::prim::ListConstruct"; - } else if (producer->kind() == torch::jit::prim::Constant) { - LOG(INFO) << "to support: torch::jit::prim::Constant"; - } else { - // aten::unbind - // prim::Constant - LOG(INFO) << "to support: some kind of producer: " << producer->kind().toQualString(); - } - return false; -} - -//TODO: this method is relatively low-level, try to use SubgraphRewriter to handle this one -//DEPRECATED -bool is_linear_if_node(torch::jit::Node* node) { - /// Check if this Node hosts a pattern like so: - /// %ret = prim::If(%1) - /// block0(): - /// %ret1 = aten::addmm(%bias, %input, %weight_t, %beta, %alpha) - /// -> (%ret1) - /// block1(): - /// %output = aten::matmul(%input, %weight_t) - /// %ret2 = aten::add(%output, %bias, %alpha) - /// -> (%ret2) - if (node->kind() != torch::jit::prim::If || node->blocks().size() != 2) { - return false; - } - - auto block2vector = [](torch::jit::Block* block, std::vector& nodes_vec) { - for (auto itr : block->nodes()) { - nodes_vec.emplace_back(itr); - } - }; - - std::vector true_nodes; - std::vector false_nodes; - block2vector(node->blocks()[0], true_nodes); - block2vector(node->blocks()[1], false_nodes); - - if (node->blocks()[0]->outputs().size() != 1 || - true_nodes.size() != 1 || - true_nodes[0]->kind() != torch::jit::aten::addmm) { - return false; - } - - if (node->blocks()[1]->outputs().size() != 1 || - false_nodes.size() != 2 || - false_nodes[0]->kind() != torch::jit::aten::matmul || - false_nodes[1]->kind() != torch::jit::aten::add ) { - return false; - } - - auto is_input_const = [](torch::jit::Node* node, int index) { - return (node->inputs()[index])->node()->kind() == torch::jit::prim::Constant; - }; - - if (true_nodes[0]->inputs().size() != 5 || - !is_input_const(true_nodes[0], 0) || - !is_input_const(true_nodes[0], 2) || - !is_input_const(true_nodes[0], 3) || - !is_input_const(true_nodes[0], 4)) { - return false; - } - - if (false_nodes[0]->inputs().size() != 2 || - !is_input_const(false_nodes[0], 1) || - false_nodes[0]->inputs()[0] != true_nodes[0]->inputs()[1] || - false_nodes[0]->inputs()[1] != true_nodes[0]->inputs()[2] || - false_nodes[1]->inputs()[0] != false_nodes[0]->outputs()[0] || - false_nodes[1]->inputs()[1] != true_nodes[0]->inputs()[0] || - false_nodes[1]->inputs()[2] != true_nodes[0]->inputs()[4]) { - return false; - } - - return true; -} - -//DEPRECATED -std::vector extract_linear_input(torch::jit::Node *node) { - std::vector valid_inputs; - if (is_linear_if_node(node)) { - valid_inputs.emplace_back(node->inputs()[0]); - auto addmm_node = *((node->blocks()[0])->nodes().begin()); - valid_inputs.emplace_back(addmm_node->inputs()[1]); - } - return valid_inputs; -} - -//DEPRECATED -bool is_dim_equal_if_node(torch::jit::Node* node) { - /// Check if this Node hosts a pattern like so: - /// %const_val : int = prim::Constant[value=2]() - /// %dim : int = aten::dim(%input_tensor) - /// %eq : bool = aten::eq(%dim, %const_val) - /// %ret = prim::If(%eq) - /// block0(): - /// ... - /// block1(): - /// ... - if (node->kind() != torch::jit::prim::If || node->blocks().size() != 2 || - node->inputs().size() != 1) { - return false; - } - - if (node->input(0)->node()->kind() == torch::jit::aten::eq) { - auto eq_node = node->input(0)->node(); - if (eq_node->inputs().size() == 2 && - eq_node->input(0)->node()->kind() == torch::jit::aten::dim && - eq_node->input(1)->node()->kind() == torch::jit::prim::Constant) { - return true; - } - } - return false; -} - -//DEPRECATED -void inline_if_body(torch::jit::Block* body) { - torch::jit::Node* owning_node = body->owningNode(); - for (auto itr = body->nodes().begin(); itr != body->nodes().end();) { - torch::jit::Node* body_node = *itr; - // advance iterator because after body_node is moved its next pointer will be to n - itr++; - body_node->moveBefore(owning_node); - } - for (size_t i = 0; i < owning_node->outputs().size(); ++i) { - owning_node->outputs().at(i)->replaceAllUsesWith(body->outputs().at(i)); - } - owning_node->destroy(); -} - -/* - bool shapeIsKnown(Value* v) { - if (v->type()->cast()) { - if (!v->isCompleteTensor()) { - return false; - } - if (*v->type()->castRaw()->dim() == 0) { - return false; - } - } - return true; - } */ - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/util/poros_util.h b/poros/poros/util/poros_util.h deleted file mode 100644 index a5ce88f1ee3..00000000000 --- a/poros/poros/util/poros_util.h +++ /dev/null @@ -1,175 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file poros_util.h -* @author tianjinjin@baidu.com -* @date Wed Apr 7 17:52:36 CST 2021 -* @brief -**/ - -#pragma once - -#include -#include -#include -#include -#include -#include - -namespace baidu { -namespace mirana { -namespace poros { - -//以下两个函数输出单个节点的信息 -inline std::string node_info_with_attr(const torch::jit::Node *n) { - std::stringstream ss; - n->print(ss, 0, {}, /**print_source_locations = */false, - /* print_attributes = */true, - /* print_scopes = */ false, - /* print_body = */ false); - return ss.str(); -} - -inline std::string node_info(const torch::jit::Node *n) { - std::stringstream ss; - n->print(ss, 0, {}, /**print_source_locations = */false, - /* print_attributes = */false, - /* print_scopes = */ false, - /* print_body = */ false); - return ss.str(); -} - -//when a node info like: %1126 : Float(*, 2, strides=[2, 1], requires_grad=0, device=cuda:0) = aten::squeeze(%output1.1, %self.new_length.1075) -//we set the output like: [aten::squeeze][ouput:%1126][input:%output1.1, %self.new_length.1075] -inline std::string layer_info(const torch::jit::Node *n) { - std::stringstream ss; - ss << "[" << n->kind().toQualString() << "]"; - auto outs = n->outputs(); - if (outs.size() > 0) { - ss << "[output:"; - size_t i = 0; - for (auto out : outs) { - if (i++ > 0) { - ss << ","; - } - ss << "%" << out->debugName(); - } - ss << "]"; - } - auto ins = n->inputs(); - if (outs.size() == 0 && ins.size() != 0) { - ss << "[][input:"; - size_t i = 0; - for (auto in : ins) { - if (i++ > 0) { - ss << ","; - } - ss << "%" << in->debugName(); - } - ss << "]"; - } - return ss.str(); -} - -// inline std::string node_info(const torch::jit::Node* n) { -// std::stringstream ss; -// ss << *n; -// std::string node_info = ss.str(); -// node_info.erase(std::remove(node_info.begin(), node_info.end(), '\n'), node_info.end()); -// return node_info; -// } - -int merge_graph_to_module(std::shared_ptr& to_merge_graph, - torch::jit::Module& module, - bool init_module_ptr); - -torch::jit::Module build_tmp_module(std::shared_ptr& sub_graph); - - -bool gen_dims_for_tensor(const torch::jit::Value* value, std::vector& dims); - -/** - * @brief update global context when some Value copy happened in the Segment progress && engine transform progress. - * 当在子图分割阶段或者后续engine转换阶段出现value的复制场景时,调用改函数,完成必要的value全局信息的拷贝。 - * 当前(2022.01)主要实现 value_dynamic_shape_map 中,value的shape信息的拷贝。 - * @param [in] old_value : 原value - * @param [in] new_value : 新的value,新value的meta信息从原value拷贝而来。 - * @return null - **/ -void update_global_context(torch::jit::Value* old_value, torch::jit::Value* new_value); - -/** - * @brief update global context when some Value copy happened in the Segment progress && engine transform progress. - * 当子图分割中出现node融合时,需要更新node维度的key - * @param [in] old_node : 原node - * @param [in] new_node : 新的node,新value的meta信息从原value拷贝而来。 - * @return null - **/ -void update_global_list_size_map_node_key_context(torch::jit::Node* old_node, torch::jit::Node* new_node); - -/** - * @brief update global context when some Value copy happened in the Segment progress && engine transform progress. - * 当子图分割中出现node融合时,需要更新node维度的key - * @param [in] old_node : 原node - * @param [in] new_node : 新的node,新value的meta信息从原value拷贝而来。 - * @return null - **/ -void update_global_int_intlist_map_context(torch::jit::Value* old_value, torch::jit::Value* new_value); - -/** - * @brief unmerge the subgraph to its parent graph(especially when engine transform failed) - * 把子图的节点信息,重新merge的父图里面去(尤其是在子图转engine失败需要fallback的场景) - * - * @param [in] subgraph_node : 类型为CudaFusionGroup的特殊node。 - * @return null - **/ -void unmerge_subgraph(torch::jit::Node* subgraph_node); - -/** - * @brief 在输入block及其子block中遍历CudaFusionGroup节点,放入to_optimized_nodes中准备优化。 - * - * @param [in] block : 需要遍历CudaFusionGroup节点的block。 - * @param [out] to_optimized_nodes : 输出遍历到的CudaFusionGroup节点集合。 - * @return null - **/ -void find_to_optimized_nodes(torch::jit::Block* block, std::vector& to_optimized_nodes); - - -/******************************************************************** - SOME DEPRECATED FUNCTIONS BELOW -*********************************************************************/ -//DEPRECATED -bool gen_dims_for_scarlar(const torch::jit::Value* value, std::vector& dims); -//DEPRECATED -bool gen_dims_for_tensorlist(const torch::jit::Value* value, std::vector& dims); - -//判断某个节点是否是可以不展开的liner节点,否则的话会展开很多的分支,把整个graph切分的过于细碎。 -//DEPRECATED -bool is_linear_if_node(torch::jit::Node* node); - -//DEPRECATED -std::vector extract_linear_input(torch::jit::Node *node); - -//判断某个if节点的输入,是否是一个aten::dim 和 const 比较生成的,如果是的,这个if节点依赖的判断条件很可能是一个常量。 -//DEPRECATED -bool is_dim_equal_if_node(torch::jit::Node* node); - -//当if的条件恒成立时,将if条件去掉,相应的block提出来。 -//DEPRECATED -void inline_if_body(torch::jit::Block* body); - -} // namespace poros -} // namespace mirana -} // namespace baidu diff --git a/poros/poros/util/test_util.cpp b/poros/poros/util/test_util.cpp deleted file mode 100644 index 0165d7ff989..00000000000 --- a/poros/poros/util/test_util.cpp +++ /dev/null @@ -1,367 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file test_util.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include "test_util.h" - -#include - -#include -#include - -#include "poros/compile/graph_prewarm.h" -#include "poros/context/poros_global.h" -#include "poros/engine/iengine.h" -#include "poros/iplugin/plugin_create.h" - -namespace baidu { -namespace mirana { -namespace poros { -namespace testutil { - -static inline void clone_tensor_vector(const std::vector &old_vec, std::vector &new_vec) { - for (auto i: old_vec) { - new_vec.push_back(i.clone()); - } -} - -static std::string get_engine_name(const baidu::mirana::poros::PorosOptions &poros_option) { - std::string engine_name(""); - if (poros_option.device == Device::CUDA) { - engine_name = "TensorrtEngine"; - } else if (poros_option.device == Device::XPU) { - engine_name = "XtclEngine"; - } else { - engine_name = ""; - } - return engine_name; -} - -static bool write_to_log(const std::string &log_path, const std::string &inform) { - if (log_path.empty()) { - return true; - } else { - std::ofstream log_file(log_path, std::ios::app); - if (log_file) { - log_file << inform; - log_file.close(); - return true; - } else { - return false; - } - } -}; - -static baidu::mirana::poros::IEngine * -select_engine(const torch::jit::Node *n, const baidu::mirana::poros::PorosOptions &poros_option) { - baidu::mirana::poros::IEngine *engine(nullptr); - if (n == nullptr || n->kind() != torch::jit::prim::CudaFusionGroup) { - return nullptr; - } - std::string engine_name = get_engine_name(poros_option); - if (engine_name.empty()) { - return nullptr; - } - engine = dynamic_cast(create_plugin(engine_name, \ - baidu::mirana::poros::PorosGlobalContext::instance()._engine_creator_map)); - if (engine == nullptr || engine->init() < 0) { - return nullptr; - } - return engine; -}; - -static bool is_node_supported(const torch::jit::Node *node, const baidu::mirana::poros::PorosOptions &poros_option) { - std::string engine_name = get_engine_name(poros_option); - auto converter_map = baidu::mirana::poros::PorosGlobalContext::instance().get_converter_map(engine_name); - if (converter_map != nullptr && converter_map->node_converterable(node)) { - LOG(INFO) << "supported node: " << node->kind().toQualString(); - return true; - } else { - if (node->kind() != torch::jit::prim::Loop && - node->kind() != torch::jit::prim::If && - node->kind() != torch::jit::prim::CudaFusionGroup) { - LOG(WARNING) << "not supported node: " << node->kind().toQualString() - << ", detail info: " << *node; - } - return false; - } -} - -static std::vector run_graph(const std::shared_ptr &graph, - const std::vector &input_data, - const baidu::mirana::poros::PorosOptions &poros_option, - const std::string &log_path) { - std::vector graph_output; - // 构建graph exe - std::string function_name("test tensor"); - torch::jit::GraphExecutor graph_exe(graph, function_name); - // 将输入导入ivalue vector - std::vector graph_input; - for (size_t i = 0; i < input_data.size(); i++) { - graph_input.push_back(input_data[i]); - } - // 执行graph - std::clock_t start, end; - if (poros_option.device == Device::CUDA) { - cudaDeviceSynchronize(); - } - start = std::clock(); - graph_exe.run(graph_input); - if (poros_option.device == Device::CUDA) { - cudaDeviceSynchronize(); - } - end = std::clock(); - std::string log_inform = "graph time:" + std::to_string(double(end - start) / CLOCKS_PER_SEC * 1000.0) + "ms\t"; - std::cout << log_inform; - if (!write_to_log(log_path, log_inform)) { - LOG(WARNING) << "write to log failed"; - } - // 提取结果 - for (size_t i = 0; i < graph_input.size(); i++) { - auto tmp_ivalue = graph_input[i]; - graph_output.push_back(tmp_ivalue.toTensor()); - } - return graph_output; -}; - -static const std::vector has_constant_tensor_inputs_node() { - return {torch::jit::aten::batch_norm, - torch::jit::aten::_convolution, - torch::jit::aten::conv1d, - torch::jit::aten::conv2d, - torch::jit::aten::conv3d, - torch::jit::aten::layer_norm, - torch::jit::aten::lstm, - torch::jit::aten::group_norm, - torch::jit::aten::instance_norm}; -} - -static std::vector run_engine(std::shared_ptr &graph, - const baidu::mirana::poros::PorosOptions &poros_option, - baidu::mirana::poros::IConverter *converter, - const std::vector &input_data, - const std::string &log_path, - const std::vector> &prewarm_data_of_engine) { - std::vector engine_output; - // 避免inplace的op,prewarm后会更改原数据,例如:add_,所以先clone - PorosGlobalContext::instance()._value_dynamic_shape_map = {}; - - PorosGlobalContext::instance().set_poros_options(poros_option); - std::vector> prewarm_clone; - for (auto it = prewarm_data_of_engine.begin(); it != prewarm_data_of_engine.end(); ++it) { - std::vector input_clone; - clone_tensor_vector(*it, input_clone); - prewarm_clone.push_back(input_clone); - } - std::vector> prewarm_datas; - for (auto it = prewarm_clone.begin(); it != prewarm_clone.end(); ++it) { - std::vector prewarm_input_data; - for (size_t i = 0; i < (*it).size(); i++) { - prewarm_input_data.push_back((*it)[i]); - } - prewarm_datas.push_back(prewarm_input_data); - } - - // 检查graph中是否包含待converter的node - bool converter_node_exist = false; - torch::jit::Node *converter_node = nullptr;; - std::string converter_node_kind_name; - for (auto node_it: graph->nodes()) { - for (auto converter_node_kind: converter->node_kind()) { // compare node kind - if (node_it->kind().toQualString() == converter_node_kind.toQualString() - && is_node_supported(node_it, poros_option)) { - converter_node_exist = true; - converter_node_kind_name = node_it->kind().toQualString(); - converter_node = node_it; - break; - } - } - } - if (!converter_node_exist) { - LOG(WARNING) << "Can't find converter node."; - return engine_output; - } - - // 判断是否是batchnormal类型 - bool convter_has_constant_tensor_inputs = false; - for (auto node_kind_it: has_constant_tensor_inputs_node()) { - if (node_kind_it.toQualString() == converter_node->kind().toQualString()) { - convter_has_constant_tensor_inputs = true; - break; - } - } - - // 插入constant tensor inputs - if (convter_has_constant_tensor_inputs) { - torch::jit::WithInsertPoint guard(graph->block()->nodes().front()); - if (converter_node->kind().toQualString() == has_constant_tensor_inputs_node()[6].toQualString()) { - auto lt = c10::List({}); - for (size_t i = 3; i < prewarm_clone[0].size(); i++) { - lt.push_back(prewarm_clone[0][i]); - } - auto lt_ivalue = c10::IValue(lt); - auto len_const = graph->insertConstant(lt_ivalue); - converter_node->replaceInput(2, len_const); - } else { - for (size_t i = 1; i < prewarm_datas[0].size(); i++) { - auto len_const = graph->insertConstant(prewarm_datas[0][i]); - if (converter_node->kind().toQualString() == has_constant_tensor_inputs_node()[5].toQualString() - || converter_node->kind().toQualString() == has_constant_tensor_inputs_node()[7].toQualString()) { - converter_node->replaceInput(i + 1, len_const); - } else { - converter_node->replaceInput(i, len_const); - } - } - } - } - // 得到预热图 - auto prewarm_graph = baidu::mirana::poros::graph_prewarm(graph, prewarm_datas); - - // 将graph全部加入subgraph - torch::jit::Node *subgraph_node = torch::jit::SubgraphUtils::createSingletonSubgraph( - *(prewarm_graph->nodes().begin()), - torch::jit::prim::CudaFusionGroup); - auto node_it = ++prewarm_graph->nodes().begin(); - while (node_it != prewarm_graph->nodes().end()) { - torch::jit::SubgraphUtils::mergeNodeIntoSubgraph(*node_it, subgraph_node); - node_it = ++prewarm_graph->nodes().begin(); - } - - // 选取与初始化engine - baidu::mirana::poros::IEngine *engine = select_engine(subgraph_node, poros_option); - if (engine == nullptr) { - LOG(WARNING) << "select engine failed"; - return engine_output; - } - - // 将graph转到engine(包括op替换) - std::shared_ptr sub_graph = subgraph_node->g(torch::jit::attr::Subgraph); - baidu::mirana::poros::PorosGraph poros_graph = {sub_graph.get(), subgraph_node}; - if (engine->transform(poros_graph) < 0) { - LOG(WARNING) << "engine transform failed"; - return engine_output; - } - - // 测试engine输出 - std::clock_t start, end; - if (poros_option.device == Device::CUDA) { - cudaDeviceSynchronize(); - } - start = std::clock(); - if (convter_has_constant_tensor_inputs) { - std::vector input_data_without_constant; - input_data_without_constant.push_back(input_data[0].clone()); - if (converter_node->kind().toQualString() == has_constant_tensor_inputs_node()[6].toQualString()) { - input_data_without_constant.push_back(input_data[1].clone()); - input_data_without_constant.push_back(input_data[2].clone()); - } - engine_output = engine->excute_engine(input_data_without_constant); - } else { - engine_output = engine->excute_engine(input_data); - } - if (poros_option.device == Device::CUDA) { - cudaDeviceSynchronize(); - } - end = std::clock(); - std::string log_inform = "engine time:" + std::to_string(double(end - start) / CLOCKS_PER_SEC * 1000.0) + "ms\t" + - converter_node_kind_name + "\n"; - - std::cout << log_inform; - - if (!write_to_log(log_path, log_inform)) { - LOG(WARNING) << "write to log failed"; - } - return engine_output; -}; - -bool run_graph_and_poros(const std::string &graph_IR, - const baidu::mirana::poros::PorosOptions &poros_option, - baidu::mirana::poros::IConverter *converter, - const std::vector &input_data, - std::vector &graph_output, - std::vector &poros_output, - const std::vector> *prewarm_data, - std::string log_path, - const std::vector const_input_indices) { - try { - // 解析graph - auto graph = std::make_shared(); - torch::jit::parseIR(graph_IR, graph.get()); - std::vector real_input; - clone_tensor_vector(input_data, real_input); - - if (!const_input_indices.empty()) { - torch::jit::WithInsertPoint guard(graph->block()->nodes().front()); - for (const size_t &index : const_input_indices) { - graph->inputs()[index]->replaceAllUsesWith(graph->insertConstant(input_data[index])); - } - for (auto it = const_input_indices.rbegin(); it != const_input_indices.rend(); it++) { - graph->eraseInput(*it); - real_input.erase(real_input.begin() + *it); - } - } - // 运行原始graph - std::vector input_of_graph; - clone_tensor_vector(real_input, input_of_graph); - graph_output = run_graph(graph, input_of_graph, poros_option, log_path); - - // convert op并运行engine - std::vector input_of_engine; - clone_tensor_vector(real_input, input_of_engine); - - // 准备预热数据 - std::vector> prewarm_data_of_engine; - if (prewarm_data == nullptr) { - prewarm_data_of_engine.push_back(input_of_engine); - } else { - for (size_t i = 0; i < (*prewarm_data).size(); ++i) { - std::vector tmp_clone_data; - clone_tensor_vector((*prewarm_data)[i], tmp_clone_data); - //讲道理,不应该出现这个情况,预防万一... - if (!const_input_indices.empty() && tmp_clone_data.size() == input_data.size()) { - for (auto it = const_input_indices.rbegin(); it != const_input_indices.rend(); it++) { - tmp_clone_data.erase(tmp_clone_data.begin() + *it); - } - } - prewarm_data_of_engine.push_back(tmp_clone_data); - } - } - poros_output = run_engine(graph, poros_option, converter, input_of_engine, log_path, prewarm_data_of_engine); - } catch (const char* err) { - LOG(ERROR) << " Exception: " << err; - return false; - } - return true; -} - - -bool almost_equal(const at::Tensor &a, const at::Tensor &b, const float &threshold) { - auto a_float = a.toType(at::kFloat); - auto b_float = b.toType(at::kFloat); - double maxValue = 0.0; - maxValue = fmax(a.abs().max().item(), maxValue); - maxValue = fmax(b.abs().max().item(), maxValue); - at::Tensor diff = a_float - b_float; - return diff.abs().max().item() <= threshold * maxValue; -} - -}// namespace testutil -}// namespace poros -}// namespace mirana -}// namespace baidu diff --git a/poros/poros/util/test_util.h b/poros/poros/util/test_util.h deleted file mode 100644 index 97ad20c9e95..00000000000 --- a/poros/poros/util/test_util.h +++ /dev/null @@ -1,70 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file test_util.h -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#pragma once - -#include "poros/compile/poros_module.h" -#include "poros/converter/iconverter.h" - -namespace baidu { -namespace mirana { -namespace poros { -namespace testutil { -/** - * @brief run graph and poros, then compare their outputs - * - * @param [in] graph_IR : string of IR - * @param [in] poros_option : default device is GPU - * @param [in] converter : converter tested - * @param [in] input_data : vector of at::Tensor, once input data of graph - * @param [in] log_path : record the running time of the graph and engine. default is none and don't record. - * @param [in] prewarm_data : preheating data, default is null and input_data is used for preheating - * @param [in] const_input_indices : the data index in input_data, which will trans to constant-tensor before graph run. - * (ie. constant weight parameter), this can change the graph and real input datas implicitly. - * @param [out] graph_output : vector of at::Tensor, graph outputs - * @param [out] poros_output : vector of at::Tensor, poros outputs - * @return bool - * @retval true => succeed false => failed -**/ -bool run_graph_and_poros(const std::string &graph_IR, - const baidu::mirana::poros::PorosOptions &poros_option, - baidu::mirana::poros::IConverter *converter, - const std::vector &input_data, - std::vector &graph_output, - std::vector &poros_output, - const std::vector> *prewarm_data = nullptr, - std::string log_path = "", - const std::vector const_input_indices = {}); - -/** - * @brief compare the similarity of two Tensors containing Float - * - * @param [in] a : first Tensor - * @param [in] b : second Tensor - * @param [in] threshold : acceptable relative threshold - * @return bool - * @retval true => succeed false => failed -**/ -bool almost_equal(const at::Tensor &a, const at::Tensor &b, const float &threshold); - -}// namespace testutil -}// namespace poros -}// namespace mirana -}// namespace baidu diff --git a/poros/python/example/example.py b/poros/python/example/example.py deleted file mode 100644 index 695c3008bf4..00000000000 --- a/poros/python/example/example.py +++ /dev/null @@ -1,95 +0,0 @@ -# Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -import os -os.environ["CUDA_VISIBLE_DEVICES"] = "0" - -import torch -import poros -import numpy as np - -def load_example_model(): - """load model示例,正常load model即可""" - import torchvision.models as models - std_resnet = models.resnet50(pretrained=True) - std_resnet.cuda() - std_resnet.eval() - return std_resnet - - -def load_example_input_datas(): - """加载预热数据""" - data_list = [] - #max size - input_1 = np.ones((3, 3, 96, 320), np.float32) - input_tensor = torch.from_numpy(input_1).cuda() - data_list.append(input_tensor) - - #min size - input_2 = np.ones((1, 3, 96, 320), np.float32) - input_tensor2 = torch.from_numpy(input_2).cuda() - data_list.append(input_tensor2) - - #opt size - input_3 = np.ones((1, 3, 96, 320), np.float32) - input_tensor3 = torch.from_numpy(input_3).cuda() - data_list.append(input_tensor3) - - return data_list - - -if __name__ == '__main__': - print("this is an example for poros") - - # step1: 按照正常的torch模块的步骤,load模型和参数,此处以resnet50为例 - # load_example_model 过程中load的原始pytorch模型(python代码),必须是完成了poros预处理的python代码 - # poros预处理相关wiki: 【待补充】 - original_model = load_example_model() - - # step2: 准备预热数据。 - # 请准备 1-3 份内容不一样的预热数据(example中是准备了3份一样的预热数据,只是示例,实际中尽量不要这样做) - # 每一份预热数据用tuple封装,除非该模型只有一个输入,且这个输入类型是torch.Tensor. - # 多份预热数据用list连接。 - # !!!注意: 预热数据是必须的。 - input_datas = load_example_input_datas() - - # step3: 调用poros,编译原始的model,得到PorosModel - # 当 option.is_dynamic 为true时,设置的预热数据的个数必须为3的倍数。 - # 当 option.is_dynamic 为false是,设置的预热数据至少为1份。 - option = poros.PorosOptions() - option.is_dynamic = True - #option.debug = True - - try: - poros_model = poros.compile(original_model, input_datas, option) - except Exception as e: - print("compile poros_model failed. error msg: {}".format(e)) - #poros_model = original_model - exit(0) - - # 序列化&反序列化 - # poros.save(poros_model, "poros_model.pt") - # poros_model = poros.load("poros_model.pt", option) - - # 准备测试用的batch数据 - input = np.ones((3, 3, 96, 320), np.float32) - batch_tensor = torch.from_numpy(input).cuda() - - # step4: 预测。 - #result = poros_model(input_datas[0]) - result = poros_model(batch_tensor) - #result = original_model(batch_tensor) - - print(result.size()) - print(result) diff --git a/poros/python/poros/__init__.py b/poros/python/poros/__init__.py deleted file mode 100644 index 0c2550fa209..00000000000 --- a/poros/python/poros/__init__.py +++ /dev/null @@ -1,34 +0,0 @@ -# Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -""" -init file for poros -""" - -import os -import sys - -if sys.version_info < (3, 6): - raise Exception("Poros can only work on Python 3.6+") - -import ctypes -import torch - -from poros._compile import * -from poros._module import PorosOptions - -def _register_with_torch(): - poros_dir = os.path.dirname(__file__) - torch.ops.load_library(poros_dir + '/lib/libporos.so') - -_register_with_torch() \ No newline at end of file diff --git a/poros/python/poros/_compile.py b/poros/python/poros/_compile.py deleted file mode 100644 index 0df9f20931f..00000000000 --- a/poros/python/poros/_compile.py +++ /dev/null @@ -1,105 +0,0 @@ -# Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -""" -compile function for poros. -""" - -from typing import List, Dict, Any -import torch -from torch import nn - -import poros._C -from poros._input_convert import convert_prewarm_inputs -from poros._input_convert import convert_poros_option -from poros._module import PorosModule - - -def wrap_cpp_module(cpp_module): - """ - Wrap torch._C.ScriptModule to porosModule, recursively for all submodules - """ - def init_fn(script_module): - """init_fn""" - for name, cpp_module in torch._C.ModuleDict(script_module._c).items(): - setattr(script_module, name, wrap_cpp_module(cpp_module)) - script_module._concrete_type = torch._C.ConcreteModuleType.from_jit_type(script_module._c._type()) - - for idx, fn in enumerate(script_module._c._get_forward_pre_hooks()): - script_module._forward_pre_hooks[idx] = fn - for idx, fn in enumerate(script_module._c._get_forward_hooks()): - script_module._forward_hooks[idx] = fn - - return PorosModule._construct(cpp_module, init_fn) - - -def load(filename, poros_options): - """ - Args: - filename( str): poros model save path - poros_options(PorosOptions / Dict of settings): compile settings for poros - Returns: - PorosModule: Compiled Module of poros, - when run it will partially execute via inlined engine (which is TensorRT) - """ - compiled_cpp_mod = poros._C.load(filename, convert_poros_option(poros_options)) - compiled_module = wrap_cpp_module(compiled_cpp_mod) - return compiled_module - -def save(module, filename): - """ - Args: - module(PorosModule): poros module - filename( str): poros model save path - """ - assert type(module).__name__ == "PorosModule", "The type of module must be PorosModule" - assert type(filename).__name__ == "str", "The type of filename must be str" - module.save(filename) - -def compile(module, prewarm_inputs, poros_options): - """ - Compile a TorchScriptModule/nn.Module to porosModule - Converts specifically the forward method of the original Module - Args: - module (torch.nn.Module / torch.jit.ScriptModule): Source module - input (list of tensor input): prewarmed data. - poros_options(PorosOptions): compile settings for poros - Returns: - PorosModule: Compiled Module of poros, - when run it will partially execute via inlined engine (which is TensorRT) - """ - if poros_options.device == "GPU": - assert "cuda" in str(list(module.state_dict().values())[0].device), \ - "If the poros_options.device is GPU, the module.device should also is GPU" - - sp_model = None - if isinstance(module, torch.jit.ScriptModule): - sp_model = module - else: - if poros_options.preprocess_mode == 0: - sp_model = torch.jit.script(module, optimize=None, _frames_up=0, _rcb=None) - elif poros_options.preprocess_mode == 1: - sp_model = torch.jit.trace(module, prewarm_inputs[0]) - else: - raise ValueError( - "preprocess_mode value err: The range of preprocess_mode is [0,1]") - - if sp_model is None: - raise TypeError( - "can't trans to poros module currently") - - wraped_inputs = convert_prewarm_inputs(prewarm_inputs) - - compiled_cpp_mod = poros._C.compile_graph(sp_model._c, wraped_inputs, convert_poros_option(poros_options)) - compiled_module = wrap_cpp_module(compiled_cpp_mod) - return compiled_module diff --git a/poros/python/poros/_input_convert.py b/poros/python/poros/_input_convert.py deleted file mode 100644 index 23eebfc5df7..00000000000 --- a/poros/python/poros/_input_convert.py +++ /dev/null @@ -1,79 +0,0 @@ -# Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -""" - convert prewarm-data to c10:ivalues list that can handle in poros. -""" - -import typing # List & Dict & Any -import torch -import poros._C - -from poros._module import DynamicOptions -from poros._module import PorosOptions -from poros._parse_util import _parse_device - -def make_to_tuple(prewarm_input): - """ - wrap a single torch.Tensor input to tuple - """ - if isinstance(prewarm_input, torch.Tensor): - return (prewarm_input,) - # done primarily so that weird iterables fail here and not pybind11 code - if not isinstance(prewarm_input, tuple): - return tuple(prewarm_input) - return prewarm_input - - -def convert_prewarm_inputs(prewarm_inputs): - # type: (Any) -> poros._C.PreWarmDatas - """ - convert prewarm-data to c10:ivalues list that can handle in poros. - we can accept 3 kinds of prewarm_inputs: - one input that has a single tensor [torch.Tensor] - one input that has multiple variables [tuple] - more than one input, that each input has a single tensor [List of torch.Tensor] - more that one input, that each input has multiple variables [List of tuple] - """ - wraped_prewarm_inputs = [] - if isinstance(prewarm_inputs, torch.Tensor): - wraped_prewarm_inputs.append(make_to_tuple(prewarm_inputs)) - elif isinstance(prewarm_inputs, tuple): - wraped_prewarm_inputs.append(prewarm_inputs) - elif isinstance(prewarm_inputs, list): - for member in prewarm_inputs: - if isinstance(member, torch.Tensor): - wraped_prewarm_inputs.append(make_to_tuple(member)) - elif isinstance(member, tuple): - wraped_prewarm_inputs.append(member) - else: - raise TypeError("prewarm_inputs for poros should be torch.Tensor or wraped as tuple, fix it") - else: - raise TypeError("prewarm_inputs for poros should be torch.Tensor or wraped as tuple or inputs-lists, fix it") - return wraped_prewarm_inputs - # info = poros._C.PreWarmDatas() - # info.set_data(prewarm_inputs) - -def convert_poros_option(poros_option): - # type: Dict[str, Any] -> poros._C.PorosOptions - """ - converter key-value poros_option to PorosOptions that can handle in poros - """ - option = poros._C.PorosOptions() - if poros_option is None: - #default situation. if user do not set the poros_option - return option - elif isinstance(poros_option, PorosOptions): - return poros_option.to_internal() - else: - raise TypeError("poros_option for poros should be PorosOptions or a attribute dict fix it") diff --git a/poros/python/poros/_module.py b/poros/python/poros/_module.py deleted file mode 100644 index 13349eaf068..00000000000 --- a/poros/python/poros/_module.py +++ /dev/null @@ -1,174 +0,0 @@ -# Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -""" -poros module here -""" - -import poros._C -from poros._parse_util import _parse_device - -from torch.jit._script import RecursiveScriptModule -from torch.jit._script import ScriptModule -from torch.jit._script import script - -class DynamicOptions(object): - """ - dynamic settings for poros - """ - def __init__(self): - """set defalut dynamic options""" - self.is_dynamic = False - self.min_shapes = [] - self.opt_shapes = [] - self.max_shapes = [] - - def set_dynamic_options(self, min, opt, max): - """situation when give three inputs is given""" - option_list = [min, opt, max] - for item in option_list: - if not (isinstance(item, list)): - raise TypeError("dynamic_option for poros should be IntList, fix it") - option_list.sort() - self.min_shapes = option_list[0] - self.opt_shapes = option_list[1] - self.max_shapes = option_list[2] - - def set_dynamic_option(self, opt): - """situation when only one input is given""" - if not isinstance(opt, list): - raise TypeError("dynamic_option for poros should be IntList, fix it") - else: - self.min_shapes = opt - self.opt_shapes = opt - self.max_shapes = opt - self.is_dynamic = False - - def get_dynamic_options(self): - """get dynamic options""" - return [self.min_shapes, self.opt_shapes, self.max_shapes] - - def to_internal(self): - """ - change DynamicOptions in python env to DynamicShapeOptions in c++ env - """ - option = poros._C.DynamicShapeOptions() - assert isinstance(self.is_dynamic, bool) - option.is_dynamic = self.is_dynamic - - assert isinstance(self.min_shapes, list) - option.min_shapes = self.min_shapes - assert isinstance(self.opt_shapes, list) - option.opt_shapes = self.opt_shapes - assert isinstance(self.max_shapes, list) - option.max_shapes = self.max_shapes - return option - -class PorosOptions(object): - """ - options for poros - """ - available_devices = ["GPU", "CPU", "XPU"] - available_debug_mode = [True, False] - def __init__(self): - self.device = "GPU" - self.debug = False - self.use_fp16 = False - self.max_workspace_size = 1 << 30 - self.is_dynamic = False - self.long_to_int = True - self.device_id = -1 - self.unconst_ops_thres = -1 - self.use_nvidia_tf32 = True - self.preprocess_mode = 0 - self.unsupport_op_list = [] - - def to_internal(self): - """ - change PorosOptions in python env to PorosOptions in c++ env - """ - option = poros._C.PorosOptions() - option.device = _parse_device(self.device) - assert isinstance(self.debug, bool) - option.debug = self.debug - assert isinstance(self.use_fp16, bool) - option.use_fp16 = self.use_fp16 - assert type(self.max_workspace_size) is int - option.max_workspace_size = self.max_workspace_size - assert isinstance(self.is_dynamic, bool) - option.is_dynamic = self.is_dynamic - assert isinstance(self.long_to_int, bool) - option.long_to_int = self.long_to_int - assert type(self.device_id) is int - option.device_id = self.device_id - assert type(self.unconst_ops_thres) is int - option.unconst_ops_thres = self.unconst_ops_thres - assert type(self.use_nvidia_tf32) is bool - option.use_nvidia_tf32 = self.use_nvidia_tf32 - assert type(self.preprocess_mode) is int - option.preprocess_mode = self.preprocess_mode - assert type(self.unsupport_op_list) is list - option.unsupport_op_list = self.unsupport_op_list - - return option - - def set_device(self, device): - """set device""" - if device not in PorosOptions.available_devices: - raise TypeError("device for poros invalid, only %s supported, fix it" % (PorosOptions.available_devices)) - self.device = device - - def set_debug(self, debug): - """set debug""" - if debug not in PorosOptions.available_debug_mode: - raise TypeError("device for poros invalid, only %s supported, fix it" % (PorosOptions.available_debug_mode)) - self.debug = debug - - -class PorosModule(RecursiveScriptModule): - """ - The core data structure of poros. - """ - def __init__(self, cpp_module): - super(PorosModule, self).__init__(cpp_module) - # self.options = PorosOptions() - # if option is not None and isinstance(option, PorosOptions): - # self.options = option - - @staticmethod - def _construct(cpp_module, init_fn): - """ - Construct a PorosModule that's ready for use. - Args: - cpp_module: The C++ Module that will hold the actual state of - this PorosModule instance. - init_fn: Lambda that initializes the PorosModule passed to it. - """ - script_module = PorosModule(cpp_module) - init_fn(script_module) - - # Finalize the ScriptModule: replace the nn.Module state with our - # custom implementations and flip the _initializing bit. - PorosModule._finalize_scriptmodule(script_module) - return script_module - - @property - def supported_engine(self): - """supported engine""" - return ["tensorrt"] - - # @property - # def options(self): - # """current options""" - # return self.options - diff --git a/poros/python/poros/_parse_util.py b/poros/python/poros/_parse_util.py deleted file mode 100644 index a3b74dc24a6..00000000000 --- a/poros/python/poros/_parse_util.py +++ /dev/null @@ -1,39 +0,0 @@ -# Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -""" - parse util for some python settings to poros c++ setting. -""" - -import typing # List & Dict & Any -import poros._C - -def _parse_device(device): - # type: Any -> poros._C.Device - """ - converter device info to Device struct that can handle in poros - """ - if isinstance(device, poros._C.Device): - return device - elif isinstance(device, str): - if device == "GPU" or device == "gpu": - return poros._C.Device.GPU - elif device == "CPU" or device == "cpu": - return poros._C.Device.CPU - elif device == "XPU" or device == "xpu": - return poros._C.Device.XPU - else: - ValueError("Got a device type unknown (type: " + str(device) + ")") - else: - raise TypeError("Device specification must be of type string or poros.Device, but got: " + - str(type(device))) \ No newline at end of file diff --git a/poros/python/poros/csrc/poros_py.cpp b/poros/python/poros/csrc/poros_py.cpp deleted file mode 100644 index 1f0f782c489..00000000000 --- a/poros/python/poros/csrc/poros_py.cpp +++ /dev/null @@ -1,102 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file poros_py.cpp -* @author tianjinjin@baidu.com -* @date Thu Jul 1 10:25:01 CST 2021 -* @brief -**/ - -#include "pybind11/pybind11.h" -#include "pybind11/stl.h" - -#include "Python.h" - -#include "torch/csrc/jit/python/pybind_utils.h" -#include "torch/csrc/utils/pybind.h" -#include "torch/custom_class.h" -#include "torch/script.h" -#include "torch/torch.h" - -#include "poros/compile/compile.h" - -namespace py = pybind11; - -namespace poros { -namespace pyapi { - -torch::jit::Module compile_graph(const torch::jit::Module& mod, - const py::list& input_list, - baidu::mirana::poros::PorosOptions& poros_option) { - auto function_schema = mod.get_method("forward").function().getSchema(); - py::gil_scoped_acquire gil; - std::vector prewarm_datas; - for (auto& input_tuple : input_list) { - torch::jit::Stack stack; - for (auto& input: input_tuple) { - stack.push_back(torch::jit::toTypeInferredIValue(input)); - } - prewarm_datas.push_back(stack); - } - - auto poros_mod = baidu::mirana::poros::CompileGraph(mod, prewarm_datas, poros_option); - if (poros_mod) { - return *poros_mod; - } else { - throw c10::Error("comile failed", ""); - } -} - -torch::jit::Module load(const std::string& filename, const baidu::mirana::poros::PorosOptions& options) { - auto poros_mod = baidu::mirana::poros::Load(filename, options); - return *poros_mod; -} - -PYBIND11_MODULE(_C, m) { - py::enum_(m, "Device", "Enum to specify device kind to build poros Module") - .value("GPU", baidu::mirana::poros::Device::CUDA, "Spiecify using GPU as the backend of poros Module") - .value("CPU", baidu::mirana::poros::Device::CPU, "Spiecify using CPU as the backend of poros Module") - .value("XPU", baidu::mirana::poros::Device::XPU, "Spiecify using XPU as the backend of poros Module") - .export_values(); - - py::class_(m, "PorosOptions") - .def(py::init<>()) - .def_readwrite("device", &baidu::mirana::poros::PorosOptions::device) - .def_readwrite("debug", &baidu::mirana::poros::PorosOptions::debug) - .def_readwrite("use_fp16", &baidu::mirana::poros::PorosOptions::use_fp16) - .def_readwrite("is_dynamic", &baidu::mirana::poros::PorosOptions::is_dynamic) - .def_readwrite("long_to_int", &baidu::mirana::poros::PorosOptions::long_to_int) - .def_readwrite("device_id", &baidu::mirana::poros::PorosOptions::device_id) - .def_readwrite("max_workspace_size", &baidu::mirana::poros::PorosOptions::max_workspace_size) - .def_readwrite("unconst_ops_thres", &baidu::mirana::poros::PorosOptions::unconst_ops_thres) - .def_readwrite("use_nvidia_tf32", &baidu::mirana::poros::PorosOptions::use_nvidia_tf32) - .def_readwrite("preprocess_mode", &baidu::mirana::poros::PorosOptions::preprocess_mode) - .def_readwrite("unsupport_op_list", &baidu::mirana::poros::PorosOptions::unsupport_op_list); - - - m.doc() = "Poros C Bindings"; - m.def( - "compile_graph", - &poros::pyapi::compile_graph, - "compile a PyTorch module and return a Poros module \ - that can significantly lower the inference latency"); - m.def( - "load", - &poros::pyapi::load, - "load poros model"); -} - -} // namespace pyapi -} // namespace poros diff --git a/poros/python/setup.py b/poros/python/setup.py deleted file mode 100644 index fb0b8cfa6a0..00000000000 --- a/poros/python/setup.py +++ /dev/null @@ -1,168 +0,0 @@ -# Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -""" -A setup module for the poros Python package. -""" - -# for python2 compatiblity -from __future__ import absolute_import -from __future__ import division -from __future__ import print_function - -import os -import glob -import sys -import setuptools -from setuptools import find_packages -from setuptools.command.develop import develop -from setuptools.command.install import install -import shutil -from torch.utils import cpp_extension -from wheel.bdist_wheel import bdist_wheel -from distutils.cmd import Command -from distutils import spawn -from distutils.sysconfig import get_python_lib -import multiprocessing - -# Constant known variables used throughout this file -THREAD_NUM = multiprocessing.cpu_count() -CXX11_ABI = False -CURRENT_PATH = os.path.dirname(os.path.abspath(__file__)) - -if "--use-cxx11-abi" in sys.argv: - sys.argv.remove("--use-cxx11-abi") - CXX11_ABI = True - -def cmake_build(): - """execute cmake build, to make the shared lib `libporos.so` """ - cwd = os.getcwd() - if spawn.find_executable('cmake') is None: - sys.stderr.write("CMake is required to build this package.\n") - sys.exit(-1) - _source_dir = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) - _build_dir = os.path.join(_source_dir, 'build') - _prefix = get_python_lib() - try: - cmake_configure_command = [ - 'cmake', - '-H{0}'.format(_source_dir), - '-B{0}'.format(_build_dir), - '-DCMAKE_INSTALL_PREFIX={0}'.format(_prefix), - ] - _generator = os.getenv('CMAKE_GENERATOR') - if _generator is not None: - cmake_configure_command.append('-G{0}'.format(_generator)) - spawn.spawn(cmake_configure_command) - spawn.spawn( - ['cmake', '--build', _build_dir, '-j', str(THREAD_NUM)]) - os.chdir(cwd) - except spawn.DistutilsExecError: - sys.stderr.write("Error while building with CMake\n") - sys.exit(-1) - -class CleanCommand(Command): - """Custom clean command to tidy up the project root.""" - PY_CLEAN_FILES = [ - './build', './dist', './poros/__pycache__', './poros/lib', './*.pyc', './*.tgz', './*.egg-info' - ] - description = "Command to tidy up the project root" - user_options = [] - - def initialize_options(self): - pass - - def finalize_options(self): - pass - - def run(self): - for path_spec in self.PY_CLEAN_FILES: - # Make paths absolute and relative to this path - abs_paths = glob.glob(os.path.normpath(os.path.join(CURRENT_PATH, path_spec))) - for path in [str(p) for p in abs_paths]: - if not path.startswith(CURRENT_PATH): - # Die if path in CLEAN_FILES is absolute + outside this directory - raise ValueError("%s is not a path inside %s" % (path, CURRENT_PATH)) - print('Removing %s' % os.path.relpath(path)) - shutil.rmtree(path) - - -if __name__ == "__main__": - """main setup function""" - poros_lib_path = os.path.join(CURRENT_PATH, "poros", "lib") - - # build libporos.so - if "clean" not in sys.argv: - cmake_build() - - if os.path.exists('./poros/lib') == False: - os.mkdir('./poros/lib') - - shutil.copy("../build/lib/libporos.so", "./poros/lib/libporos.so") - - # this is for torch customer extension - C = cpp_extension.CppExtension( - 'poros._C', [ - 'poros/csrc/poros_py.cpp', - ], - library_dirs=[poros_lib_path, "../third_party/tensorrtlib"], - libraries=["poros"], - include_dirs=[ - CURRENT_PATH + "/poros/csrc", - CURRENT_PATH + "/../build/include", - - ], - extra_compile_args=[ - "-Wno-deprecated", - "-Wno-deprecated-declarations", - "-Wno-unused-function", - '-Werror', - '-fopenmp', - '-D__const__=', '-g', '-O2', '-fPIC', - ], - extra_link_args=[ - "-Wno-deprecated", "-Wno-deprecated-declarations", - "-Wno-unused-function", - "-Wl,--no-as-needed", - "-lporos", - "-lnvinfer", - "-lnvinfer_plugin", - "-Wl,-rpath,$ORIGIN/lib", - "-lpthread", "-ldl", "-lutil", "-lrt", "-lm", "-Xlinker", "-export-dynamic", - ], - ) - - setuptools.setup( - name="poros", - version="0.1.0", - author="PorosTeam@BaiDu", - description='A compiler backend for PyTorch and automatically accelerate inference using tensorrt engine', - ext_modules=[C], - packages=find_packages(), - include_package_data=True, - package_data={ - 'poros': ['lib/*.so'], - }, - exclude_package_data={ - '': ['*.cpp', '*.h'], - 'poros': ['csrc/*.cpp'], - }, - install_requires=[ - 'torch>=1.9.0', - ], - cmdclass={ - 'clean': CleanCommand, - 'build_ext': cpp_extension.BuildExtension, - }, - ) - print("Setup baidu.poros python module success!") diff --git a/poros/third_party/gflags b/poros/third_party/gflags deleted file mode 160000 index a738fdf9338..00000000000 --- a/poros/third_party/gflags +++ /dev/null @@ -1 +0,0 @@ -Subproject commit a738fdf9338412f83ab3f26f31ac11ed3f3ec4bd diff --git a/poros/third_party/googletest b/poros/third_party/googletest deleted file mode 160000 index 7a7231c4424..00000000000 --- a/poros/third_party/googletest +++ /dev/null @@ -1 +0,0 @@ -Subproject commit 7a7231c442484be389fdf01594310349ca0e42a8 diff --git a/poros/tools/main.cpp b/poros/tools/main.cpp deleted file mode 100644 index a8fc811bba2..00000000000 --- a/poros/tools/main.cpp +++ /dev/null @@ -1,158 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file main.cpp -* @author tianjinjin@baidu.com -* @date Tue Mar 9 14:43:42 CST 2021 -* @brief a tool to change original serialized script module to an optimized one -**/ - -#include -#include -#include -#include -#include - -#include "poros/compile/compile.h" -//#include "poros/compile/poros_module.h" - -DEFINE_int32(batch_size, -1, "the batch size for model inference"); -DEFINE_int32(repeated_num, -1000, "how many repeated test times for a single input data"); -DEFINE_string(test_mode, -"poros", "which module we test this time: that are only three option: poros/original"); -DEFINE_string(module_file_path, -"../model/std_pretrained_resnet50_gpu.pt", "the model file path, replace this with a real one"); -DEFINE_bool(is_dynamic, -false, "the model type, used to choose input data"); - -void build_test_data(int batch_size, - std::vector> &prewarm_datas, bool is_dynamic) { - - - - std::vector inputs; - - if (is_dynamic == false) { - inputs.push_back(at::randn({ batch_size, 3, 224, 224}, {at::kCUDA})); - prewarm_datas.push_back(inputs); - return; - } - //max - inputs.push_back(at::randn({16, 3, 224, 224}, {at::kCUDA})); - prewarm_datas.push_back(inputs); - //min - std::vector inputs2; - inputs2.push_back(at::randn({1, 3, 224, 224}, {at::kCUDA})); - prewarm_datas.push_back(inputs2); - - //opt - std::vector inputs3; - inputs3.push_back(at::randn({6, 3, 224, 224}, {at::kCUDA})); - prewarm_datas.push_back(inputs3); - -} - -/* load a serialized script module and optimize it -this run as a convertion tool */ -int main(int argc, char *argv[]) { - google::ParseCommandLineFlags(&argc, &argv, true); -// gflags::SetCommandLineOption("flagfile", "./conf/gflags.conf"); - - torch::jit::Module mod; - struct timeval start, end; - float time_use; - ////////////////////////////////////////////////////////////////// - //step1: load the origin model file - ////////////////////////////////////////////////////////////////// - try { - // Deserialize the ScriptModule from a file using torch::jit::load(). - mod = torch::jit::load(FLAGS_module_file_path); - } catch (const c10::Error &e) { - std::cerr << "error loading the model\n" << e.msg(); - return -1; - } - mod.eval(); - //mod.to(at::kCPU); - mod.to(at::kCUDA); - ////////////////////////////////////////////////////////////////// - //step2: prepare input data - ////////////////////////////////////////////////////////////////// - // Create a vector of inputs for std-resnet50. - std::vector > prewarm_datas; - build_test_data(FLAGS_batch_size, prewarm_datas, FLAGS_is_dynamic); - //mod.forward(prewarm_datas[0]); - - std::cout << "input data is ok: prewarm_datas size: " << prewarm_datas.size() << std::endl; - - ////////////////////////////////////////////////////////////////// - //step3: change mode according to given test mode and press - ////////////////////////////////////////////////////////////////// - int warm_up_cycle = 50; - if (FLAGS_test_mode == "poros") { - baidu::mirana::poros::PorosOptions option; - option.device = baidu::mirana::poros::Device::CUDA; - option.is_dynamic = FLAGS_is_dynamic; - option.debug = true; - - auto poros_mod = baidu::mirana::poros::Compile(mod, prewarm_datas, option); - //poros_mod->to(at::kCUDA); - torch::jit::getProfilingMode() = true; - torch::jit::getExecutorMode() = true; - torch::jit::setGraphExecutorOptimize(false); - - //warm up - for (int i = 1; i < warm_up_cycle; i++) { - poros_mod->forward(prewarm_datas[0]); - } - //real press func - gettimeofday(&start, NULL); - for (int i = 1; i < FLAGS_repeated_num; i++) { - auto output = poros_mod->forward(prewarm_datas[0]); - } - gettimeofday(&end, NULL); - - } else if (FLAGS_test_mode == "original") { - - GRAPH_DUMP("graph info:", mod.get_method("forward").graph()); - //warm up - for (int i = 1; i < warm_up_cycle; i++) { - mod.forward(prewarm_datas[0]); - } - //real press func - gettimeofday(&start, NULL); - for (int i = 1; i < FLAGS_repeated_num; i++) { - auto output = mod.forward(prewarm_datas[0]); - } - gettimeofday(&end, NULL); - GRAPH_DUMP("torch.jit.last_executed_optimized_graph", torch::jit::lastExecutedOptimizedGraph()); - } else { - std::cerr << "given test module info: " << FLAGS_test_mode.c_str() << " not supported" - << ", only poros/original supported"; - return -1; - } - - ////////////////////////////////////////////////////////////////// - //step4: print press result - ////////////////////////////////////////////////////////////////// - time_use = (end.tv_sec - start.tv_sec) + (end.tv_usec - start.tv_usec) / (double) 1000000; - std::cout << "press mode: " << FLAGS_test_mode.c_str() - << ", repeted times: " << FLAGS_repeated_num - << ", spend time: " << time_use / FLAGS_repeated_num * 1000 - << " ms/infer" << std::endl; - - std::cout << "test done QAQ\n"; -} diff --git a/poros/unittest/CMakeLists.txt b/poros/unittest/CMakeLists.txt deleted file mode 100644 index 4c5f3034663..00000000000 --- a/poros/unittest/CMakeLists.txt +++ /dev/null @@ -1,41 +0,0 @@ -cmake_minimum_required(VERSION 3.21) -project(unittest) -set(CMAKE_CXX_STANDARD 14) - -enable_testing() -include(GoogleTest) - -set(GRAPHTEST "graph_test" ) - -file( - GLOB UT_FILES - "./op_fuser/*.cpp" - "../poros/lowering/fuse_*.cpp" -) -list(APPEND UT_FILES - "../poros/lowering/op_fuse_pass.cpp" - "../poros/util/graph_test_helper.cpp") - -add_executable(${GRAPHTEST} ${UT_FILES}) -target_link_libraries(${GRAPHTEST} gtest_main) -target_link_libraries(${GRAPHTEST} gflags::gflags) -#target_link_libraries(${UNITTEST} TensorRT::TensorRT) -target_link_libraries(${GRAPHTEST} torch) -#target_link_libraries(${UNITTEST} CUDA::cudart CUDA::cusolver CUDA::cublas CUDA::cusolver CUDA::cusparse) - -# unit test -set(UNITTEST "unit_test" ) - -file( - GLOB UT_FILES - "../poros/*/*.cpp" - "../poros/converter/*/*.cpp" - "./converter/*.cpp" -) - -add_executable(${UNITTEST} ${UT_FILES}) -target_link_libraries(${UNITTEST} gtest_main) -target_link_libraries(${UNITTEST} gflags::gflags) -target_link_libraries(${UNITTEST} TensorRT::TensorRT TensorRT::Plugin) -target_link_libraries(${UNITTEST} torch) -target_link_libraries(${UNITTEST} CUDA::cudart CUDA::cusolver CUDA::cublas CUDA::cusolver CUDA::cusparse) diff --git a/poros/unittest/converter/activation_test.cpp b/poros/unittest/converter/activation_test.cpp deleted file mode 100644 index ab6a16bfbbb..00000000000 --- a/poros/unittest/converter/activation_test.cpp +++ /dev/null @@ -1,228 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file activation_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/activation.h" -#include "poros/util/test_util.h" - -static void activation_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - std::vector shape1 = {5}, - bool single_input = true, - std::vector shape2 = {5}){ - std::vector input_data; - input_data.push_back(at::randn(shape1, {at::kCUDA})); - if (!single_input){ - input_data.push_back(at::randn(shape2, {at::kCUDA})); - } - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - std::string gelu_node_str("aten::gelu"); - if (converter->node_kind()[0].toQualString() == gelu_node_str){ - // NOTE: The official tensorrt plugin applies the Gelu activation x * Phi(x), where Phi is the Gaussian cdf, - // approximated by: 0.5 * (1 + tanh(sqrt(2 / M_PI) * (x + 0.044715 * x^3))) and the pytorch uses - // c10::cuda::compat::normcdf to compute Phi(x). So there's a difference here and therefore the threshold is slightly - // higher than other ops. One in ten runs will give you an out of normal threshold result. - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 5e-2)); - }else{ - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); - } -} - -static std::string gen_single_node_graph(const std::string& op) { - return R"IR( - graph(%0 : Tensor): - %1 : Tensor = aten::)IR" + op + R"IR((%0) - return (%1))IR"; -} - -static std::string gen_hardtanh_graph(const std::string& op, - const std::string& min_val, - const std::string& max_val) { - return R"IR( - graph(%0 : Tensor): - %1 : float = prim::Constant[value=)IR" + min_val + R"IR(]() - %2 : float = prim::Constant[value=)IR" + max_val + R"IR(]() - %3 : Tensor = aten::)IR" + op + R"IR((%0, %1, %2) - return (%3))IR"; -} -static std::string gen_leakyrelu_graph(const std::string& op, const std::string& negative_slope) { - return R"IR( - graph(%0 : Tensor): - %1 : float = prim::Constant[value=)IR" + negative_slope + R"IR(]() - %2 : Tensor = aten::)IR" + op + R"IR((%0, %1) - return (%2))IR"; -} - -static std::string gen_elu_graph(const std::string& alpha) { - return R"IR( - graph(%0 : Tensor): - %1 : float = prim::Constant[value=)IR" + alpha + R"IR(]() - %2 : int = prim::Constant[value=1]() - %3 : Tensor = aten::elu(%0, %1, %2, %2) - return (%3))IR"; -} - -TEST(Converters, ATenReluConvertsCorrectly) { - // aten::relu(Tensor self) -> Tensor - const auto graph_IR = gen_single_node_graph("relu"); - baidu::mirana::poros::ActivationConverter activationconverter; - activation_test_helper(graph_IR, &activationconverter); -} - -TEST(Converters, ATenRelu_ConvertsCorrectly) { - // aten::relu_(Tensor(a!) self) -> Tensor(a!) - const auto graph_IR = gen_single_node_graph("relu_"); - baidu::mirana::poros::ActivationConverter activationconverter; - activation_test_helper(graph_IR, &activationconverter); -} - -TEST(Converters, ATenRelu6ConvertsCorrectly) { - // aten::relu6(Tensor self) -> Tensor - const auto graph_IR = gen_single_node_graph("relu6"); - baidu::mirana::poros::ActivationConverter activationconverter; - activation_test_helper(graph_IR, &activationconverter); -} - -TEST(Converters, ATenRelu6_ConvertsCorrectly) { - // aten::relu6_(Tensor(a!) self) -> Tensor(a!) - const auto graph_IR = gen_single_node_graph("relu6_"); - baidu::mirana::poros::ActivationConverter activationconverter; - activation_test_helper(graph_IR, &activationconverter); -} - -TEST(Converters, ATenSigmoidConvertsCorrectly) { - // aten::sigmoid(Tensor self) -> Tensor - const auto graph_IR = gen_single_node_graph("sigmoid"); - baidu::mirana::poros::ActivationConverter activationconverter; - activation_test_helper(graph_IR, &activationconverter); -} - -TEST(Converters, ATenSigmoid_ConvertsCorrectly) { - // aten::sigmoid_(Tensor(a!) self) -> Tensor(a!) - const auto graph_IR = gen_single_node_graph("sigmoid_"); - baidu::mirana::poros::ActivationConverter activationconverter; - activation_test_helper(graph_IR, &activationconverter); -} - -TEST(Converters, ATenTanhConvertsCorrectly) { - // aten::tanh(Tensor self) -> Tensor" - const auto graph_IR = gen_single_node_graph("tanh"); - baidu::mirana::poros::ActivationConverter activationconverter; - activation_test_helper(graph_IR, &activationconverter); -} - -TEST(Converters, ATenTanh_ConvertsCorrectly) { - // aten::tanh_(Tensor(a!) self) -> Tensor(a!) - const auto graph_IR = gen_single_node_graph("tanh_"); - baidu::mirana::poros::ActivationConverter activationconverter; - activation_test_helper(graph_IR, &activationconverter); -} - -TEST(Converters, ATenGeluConvertsCorrectly) { - std::string graph_IR_str; - // aten::gelu schema changed in torch-1.12 - if (TORCH_VERSION_MAJOR < 2 && TORCH_VERSION_MINOR < 12) { - // aten::gelu(Tensor self) -> Tensor - graph_IR_str = gen_single_node_graph("gelu"); - } else { - // aten::gelu(Tensor self, *, str approximate='none') -> Tensor - graph_IR_str = R"IR( - graph(%0 : Tensor): - %approximate : str = prim::Constant[value="tanh"]() - %1 : Tensor = aten::gelu(%0, %approximate) - return (%1))IR"; - } - const auto graph_IR = graph_IR_str; - baidu::mirana::poros::GeluActivationConverter geluactivationconverter; - activation_test_helper(graph_IR, &geluactivationconverter, {10}); -} - -TEST(Converters, ATenLeakyreluConvertsCorrectly) { - // aten::leaky_relu(Tensor self, Scalar negative_slope=0.01) -> Tensor - const auto graph_IR = gen_leakyrelu_graph("leaky_relu", "0.01"); - baidu::mirana::poros::ActivationConverter activationconverter; - activation_test_helper(graph_IR, &activationconverter); -} - -TEST(Converters, ATenLeakyreluNegSlopeConvertsCorrectly) { - // aten::leaky_relu(Tensor self, Scalar negative_slope=0.01) -> Tensor - const auto graph_IR = gen_leakyrelu_graph("leaky_relu", "0.05"); - baidu::mirana::poros::ActivationConverter activationconverter; - activation_test_helper(graph_IR, &activationconverter); -} - -TEST(Converters, ATenHardtanhConvertsCorrectly) { - // aten::hardtanh(Tensor self, Scalar min_val=-1, Scalar max_val=1) -> Tensor - const auto graph_IR = gen_hardtanh_graph("hardtanh", "-1.0", "1.0"); - baidu::mirana::poros::ActivationConverter activationconverter; - activation_test_helper(graph_IR, &activationconverter); -} - -TEST(Converters, ATenHardtanhMinvalMaxvalConvertsCorrectly) { - // aten::hardtanh(Tensor self, Scalar min_val=-1, Scalar max_val=1) -> Tensor - const auto graph_IR = gen_hardtanh_graph("hardtanh", "-3.5", "2.5"); - baidu::mirana::poros::ActivationConverter activationconverter; - activation_test_helper(graph_IR, &activationconverter); -} - -TEST(Converters, ATenHardtanh_ConvertsCorrectly) { - // aten::hardtanh_(Tensor(a!) self, Scalar min_val=-1, Scalar max_val=1) -> Tensor(a!) - const auto graph_IR = gen_hardtanh_graph("hardtanh_", "-1.0", "1.0"); - baidu::mirana::poros::ActivationConverter activationconverter; - activation_test_helper(graph_IR, &activationconverter); -} - -TEST(Converters, ATenHardtanh_MinvalMaxvalConvertsCorrectly) { - // aten::hardtanh_(Tensor(a!) self, Scalar min_val=-1, Scalar max_val=1) -> Tensor(a!) - const auto graph_IR = gen_hardtanh_graph("hardtanh_", "-2.1", "3.8"); - baidu::mirana::poros::ActivationConverter activationconverter; - activation_test_helper(graph_IR, &activationconverter); -} - -TEST(Converters, ATenEluConvertsCorrectly) { - // aten::elu(Tensor self, Scalar alpha=1, Scalar scale=1, Scalar input_scale=1) -> Tensor - const auto graph_IR = gen_elu_graph("1.0"); - baidu::mirana::poros::ActivationConverter activationconverter; - activation_test_helper(graph_IR, &activationconverter); -} - -TEST(Converters, ATenEluAlphaConvertsCorrectly) { - // aten::elu(Tensor self, Scalar alpha=1, Scalar scale=1, Scalar input_scale=1) -> Tensor - const auto graph_IR = gen_elu_graph("3.4"); - baidu::mirana::poros::ActivationConverter activationconverter; - activation_test_helper(graph_IR, &activationconverter); -} - -TEST(Converters, ATenEluNegAlphaConvertsCorrectly) { - // aten::elu(Tensor self, Scalar alpha=1, Scalar scale=1, Scalar input_scale=1) -> Tensor - const auto graph_IR = gen_elu_graph("-2.1"); - baidu::mirana::poros::ActivationConverter activationconverter; - activation_test_helper(graph_IR, &activationconverter); -} \ No newline at end of file diff --git a/poros/unittest/converter/add_test.cpp b/poros/unittest/converter/add_test.cpp deleted file mode 100644 index 093be5da7e3..00000000000 --- a/poros/unittest/converter/add_test.cpp +++ /dev/null @@ -1,441 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file add_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/add.h" -#include "poros/util/test_util.h" - -static void add_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - bool singleInput, - std::vector shape1 = {5}, - std::vector shape2 = {5}){ - std::vector input_data; - input_data.push_back(at::randn(shape1, {at::kCUDA})); - if (!singleInput){ - input_data.push_back(at::randn(shape2, {at::kCUDA})); - } - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -static std::string gen_add_sub_tensor_graph(const std::string& op, - const std::string& alpha) { - return R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : float = prim::Constant[value=)IR" + alpha + R"IR(]() - %3 : Tensor = aten::)IR" + op + R"IR((%0, %1, %2) - return (%3))IR"; -} - -static std::string gen_add_sub_scalar_graph(const std::string& op, - const std::string& scalar, - const std::string& alpha) { - return R"IR( - graph(%0 : Tensor): - %1 : float = prim::Constant[value=)IR" + scalar + R"IR(]() - %2 : float = prim::Constant[value=)IR" + alpha + R"IR(]() - %3 : Tensor = aten::)IR" + op + R"IR((%0, %1, %2) - return (%3))IR"; -} - -TEST(Converters, ATenAddTensorConvertsCorrectly) { - // aten::add.Tensor(Tensor self, Tensor other, Scalar alpha=1) -> Tensor - const auto graph_IR = gen_add_sub_tensor_graph("add", "1.0"); - baidu::mirana::poros::AddConverter addconverter; - add_test_helper(graph_IR, &addconverter, false); - add_test_helper(graph_IR, &addconverter, false, {3, 4}, {4}); - add_test_helper(graph_IR, &addconverter, false, {4}, {3, 4}); - add_test_helper(graph_IR, &addconverter, false, {4, 1}, {1, 4}); - add_test_helper(graph_IR, &addconverter, false, {3, 4, 3}, {4, 3}); - add_test_helper(graph_IR, &addconverter, false, {4, 3}, {3, 4, 3}); -} - -TEST(Converters, ATenAddScalarConvertsCorrectly) { - // aten::add.Scalar(Tensor self, Scalar other, Scalar alpha=1) -> Tensor - const auto graph_IR = gen_add_sub_scalar_graph("add", "2.2", "1.0"); - baidu::mirana::poros::AddConverter addconverter; - add_test_helper(graph_IR, &addconverter, true); - add_test_helper(graph_IR, &addconverter, true, {3, 4, 3}); -} - -TEST(Converters, ATenAdd_TensorConvertsCorrectly) { - // aten::add_.Tensor(Tensor(a!) self, Tensor other, *, Scalar alpha=1) -> Tensor(a!) - const auto graph_IR = gen_add_sub_tensor_graph("add_", "1.0"); - baidu::mirana::poros::AddConverter addconverter; - add_test_helper(graph_IR, &addconverter, false); - add_test_helper(graph_IR, &addconverter, false, {3, 4}, {4}); - add_test_helper(graph_IR, &addconverter, false, {3, 4, 3}, {4, 3}); -} - -TEST(Converters, ATenAdd_ScalarConvertsCorrectly) { - // aten::add_.Scalar(Tensor(a!) self, Scalar other, Scalar alpha=1) -> Tensor(a!) - const auto graph_IR = gen_add_sub_scalar_graph("add_", "2.2", "1.0"); - baidu::mirana::poros::AddConverter addconverter; - add_test_helper(graph_IR, &addconverter, true); - add_test_helper(graph_IR, &addconverter, true, {3, 4, 3}); -} - -TEST(Converters, ATenAddTensorAlphaConvertsCorrectly) { - // aten::add.Tensor(Tensor self, Tensor other, Scalar alpha=1) -> Tensor - const auto graph_IR = gen_add_sub_tensor_graph("add", "2.5"); - baidu::mirana::poros::AddConverter addconverter; - add_test_helper(graph_IR, &addconverter, false); - add_test_helper(graph_IR, &addconverter, false, {3, 4}, {4}); - add_test_helper(graph_IR, &addconverter, false, {4}, {3, 4}); - add_test_helper(graph_IR, &addconverter, false, {4, 1}, {1, 4}); - add_test_helper(graph_IR, &addconverter, false, {3, 4, 3}, {4, 3}); - add_test_helper(graph_IR, &addconverter, false, {4, 3}, {3, 4, 3}); -} - -TEST(Converters, ATenAddScalarAlphaConvertsCorrectly) { - // aten::add.Scalar(Tensor(a!) self, Scalar other, Scalar alpha=1) -> Tensor(a!) - const auto graph_IR = gen_add_sub_scalar_graph("add", "2.2", "2.5"); - baidu::mirana::poros::AddConverter addconverter; - add_test_helper(graph_IR, &addconverter, true); - add_test_helper(graph_IR, &addconverter, true, {3, 4, 3}); -} - -TEST(Converters, ATenAdd_TensorAlphaConvertsCorrectly) { - // aten::add_.Tensor(Tensor self, Tensor other, Scalar alpha=1) -> Tensor - const auto graph_IR = gen_add_sub_tensor_graph("add_", "2.5"); - baidu::mirana::poros::AddConverter addconverter; - add_test_helper(graph_IR, &addconverter, false); - add_test_helper(graph_IR, &addconverter, false, {3, 4}, {4}); - add_test_helper(graph_IR, &addconverter, false, {3, 4, 3}, {4, 3}); -} - -TEST(Converters, ATenAdd_ScalarAlphaConvertsCorrectly) { - // aten::add_.Scalar(Tensor(a!) self, Scalar other, Scalar alpha=1) -> Tensor(a!) - const auto graph_IR = gen_add_sub_scalar_graph("add_", "2.2", "2.5"); - baidu::mirana::poros::AddConverter addconverter; - add_test_helper(graph_IR, &addconverter, true); - add_test_helper(graph_IR, &addconverter, true, {3, 4, 3}); -} - -TEST(Converters, ATenSubTensorConvertsCorrectly) { - // aten::sub.Tensor(Tensor self, Tensor other, Scalar alpha=1) -> Tensor - const auto graph_IR = gen_add_sub_tensor_graph("sub", "1.0"); - baidu::mirana::poros::SubConverter subconverter; - add_test_helper(graph_IR, &subconverter, false); - add_test_helper(graph_IR, &subconverter, false, {3, 4}, {4}); - add_test_helper(graph_IR, &subconverter, false, {4}, {3, 4}); - add_test_helper(graph_IR, &subconverter, false, {4, 1}, {1, 4}); - add_test_helper(graph_IR, &subconverter, false, {3, 4, 3}, {4, 3}); - add_test_helper(graph_IR, &subconverter, false, {4, 3}, {3, 4, 3}); -} - -TEST(Converters, ATenSubScalarConvertsCorrectly) { - // aten::sub.Scalar(Tensor self, Scalar other, Scalar alpha=1) -> Tensor - const auto graph_IR = gen_add_sub_scalar_graph("sub", "2.2", "1.0"); - baidu::mirana::poros::SubConverter subconverter; - add_test_helper(graph_IR, &subconverter, true); - add_test_helper(graph_IR, &subconverter, true, {3, 4, 3}); -} - -TEST(Converters, ATenSub_TensorConvertsCorrectly) { - // aten::sub_.Tensor(Tensor(a!) self, Tensor other, *, Scalar alpha=1) -> Tensor(a!) - const auto graph_IR = gen_add_sub_tensor_graph("sub_", "1.0"); - baidu::mirana::poros::SubConverter subconverter; - add_test_helper(graph_IR, &subconverter, false); - add_test_helper(graph_IR, &subconverter, false, {3, 4}, {4}); - add_test_helper(graph_IR, &subconverter, false, {3, 4, 3}, {4, 3}); -} - -TEST(Converters, ATenSub_ScalarConvertsCorrectly) { - // aten::sub_.Scalar(Tensor(a!) self, Scalar other, Scalar alpha=1) -> Tensor(a!) - const auto graph_IR = gen_add_sub_scalar_graph("sub_", "2.2", "1.0"); - baidu::mirana::poros::SubConverter subconverter; - add_test_helper(graph_IR, &subconverter, true); - add_test_helper(graph_IR, &subconverter, true, {3, 4, 3}); -} - -TEST(Converters, ATenSubTensorAlphaConvertsCorrectly) { - // aten::sub.Tensor(Tensor self, Tensor other, Scalar alpha=1) -> Tensor - const auto graph_IR = gen_add_sub_tensor_graph("sub", "2.5"); - baidu::mirana::poros::SubConverter subconverter; - add_test_helper(graph_IR, &subconverter, false); - add_test_helper(graph_IR, &subconverter, false, {3, 4}, {4}); - add_test_helper(graph_IR, &subconverter, false, {4}, {3, 4}); - add_test_helper(graph_IR, &subconverter, false, {4, 1}, {1, 4}); - add_test_helper(graph_IR, &subconverter, false, {3, 4, 3}, {4, 3}); - add_test_helper(graph_IR, &subconverter, false, {4, 3}, {3, 4, 3}); -} - -TEST(Converters, ATenSubScalarAlphaConvertsCorrectly) { - // aten::sub.Scalar(Tensor(a!) self, Scalar other, Scalar alpha=1) -> Tensor(a!) - const auto graph_IR = gen_add_sub_scalar_graph("sub", "2.2", "2.5"); - baidu::mirana::poros::SubConverter subconverter; - add_test_helper(graph_IR, &subconverter, true); - add_test_helper(graph_IR, &subconverter, true, {3, 4, 3}); -} - -TEST(Converters, ATenSub_TensorAlphaConvertsCorrectly) { - // aten::sub_.Tensor(Tensor self, Tensor other, Scalar alpha=1) -> Tensor - const auto graph_IR = gen_add_sub_tensor_graph("sub_", "2.5"); - baidu::mirana::poros::SubConverter subconverter; - add_test_helper(graph_IR, &subconverter, false); - add_test_helper(graph_IR, &subconverter, false, {3, 4}, {4}); - add_test_helper(graph_IR, &subconverter, false, {3, 4, 3}, {4, 3}); -} - -TEST(Converters, ATenSub_ScalarAlphaConvertsCorrectly) { - // aten::sub_.Scalar(Tensor(a!) self, Scalar other, Scalar alpha=1) -> Tensor(a!) - const auto graph_IR = gen_add_sub_scalar_graph("sub_", "2.2", "2.5"); - baidu::mirana::poros::SubConverter subconverter; - add_test_helper(graph_IR, &subconverter, true); - add_test_helper(graph_IR, &subconverter, true, {3, 4, 3}); -} - -TEST(Converters, ATenRsubTensorConvertsCorrectly) { - // aten::rsub.Tensor(Tensor self, Tensor other, Scalar alpha=1) -> (Tensor) - const auto graph_IR = gen_add_sub_tensor_graph("rsub", "1.0"); - baidu::mirana::poros::RsubConverter rsubconverter; - add_test_helper(graph_IR, &rsubconverter, false); - add_test_helper(graph_IR, &rsubconverter, false, {3, 4}, {4}); - add_test_helper(graph_IR, &rsubconverter, false, {4}, {3, 4}); - add_test_helper(graph_IR, &rsubconverter, false, {4, 1}, {1, 4}); - add_test_helper(graph_IR, &rsubconverter, false, {3, 4, 3}, {4, 3}); - add_test_helper(graph_IR, &rsubconverter, false, {4, 3}, {3, 4, 3}); -} - -TEST(Converters, ATenRsubScalarConvertsCorrectly) { - // aten::rsub.Scalar(Tensor self, Scalar other, Scalar alpha=1) -> (Tensor) - const auto graph_IR = gen_add_sub_scalar_graph("rsub", "2.2", "1.0"); - baidu::mirana::poros::RsubConverter rsubconverter; - add_test_helper(graph_IR, &rsubconverter, true); - add_test_helper(graph_IR, &rsubconverter, true, {3, 4, 3}); -} - -TEST(Converters, ATenRsubTensorAlphaConvertsCorrectly) { - // aten::rsub.Tensor(Tensor self, Tensor other, Scalar alpha=1) -> (Tensor) - const auto graph_IR = gen_add_sub_tensor_graph("rsub", "3.33"); - baidu::mirana::poros::RsubConverter rsubconverter; - add_test_helper(graph_IR, &rsubconverter, false); - add_test_helper(graph_IR, &rsubconverter, false, {3, 4}, {4}); - add_test_helper(graph_IR, &rsubconverter, false, {4}, {3, 4}); - add_test_helper(graph_IR, &rsubconverter, false, {4, 1}, {1, 4}); - add_test_helper(graph_IR, &rsubconverter, false, {3, 4, 3}, {4, 3}); - add_test_helper(graph_IR, &rsubconverter, false, {4, 3}, {3, 4, 3}); -} - -TEST(Converters, ATenRsubScalarAlphaConvertsCorrectly) { - // aten::rsub.Scalar(Tensor self, Scalar other, Scalar alpha=1) -> (Tensor) - const auto graph_IR = gen_add_sub_scalar_graph("rsub", "2.2", "4.44"); - baidu::mirana::poros::RsubConverter rsubconverter; - add_test_helper(graph_IR, &rsubconverter, true); - add_test_helper(graph_IR, &rsubconverter, true, {3, 4, 3}); -} - -TEST(Converters, ATenRsubTensorTypePromotionConvertsCorrectly) { - // aten::rsub.Tensor(Tensor self, Tensor other, Scalar alpha=1) -> (Tensor) - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : float = prim::Constant[value=3.33]() - %3 : Tensor = aten::rsub(%0, %1, %2) - return (%3))IR"; - baidu::mirana::poros::RsubConverter rsubconverter; - - std::vector input_data; - input_data.push_back(at::randn({3,4,3}, {at::kCUDA})); - input_data.push_back(at::ones({3,4,3}, {at::kCUDA}).to(at::ScalarType::Int)); - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &rsubconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenRsubScalarTypePromotionConvertsCorrectly) { - // aten::rsub.Tensor(Tensor self, Tensor other, Scalar alpha=1) -> (Tensor) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=5]() - %2 : float = prim::Constant[value=3.33]() - %3 : Tensor = aten::rsub(%0, %1, %2) - return (%3))IR"; - baidu::mirana::poros::RsubConverter rsubconverter; - add_test_helper(graph_IR, &rsubconverter, true); -} - -static void add_sub_dynamic_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - const std::vector& input_data, - bool is_dynamic = false, - std::vector>* prewarm_data = nullptr) { - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = is_dynamic; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output, prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenAddIntdynamicConvertsCorrectly) { - // aten::add.int(int a, int b) -> (int) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=0]() - %2 : int = prim::Constant[value=1]() - %3 : int = aten::size(%0, %1) - %4 : int = aten::size(%0, %2) - %5 : int = aten::add(%3, %4) - %6 : Tensor = aten::add(%0, %5, %2) - return (%6))IR"; - baidu::mirana::poros::AddConverter addconverter; - std::vector input_data; - input_data.push_back(at::zeros({2, 3}, {at::kCUDA}).to(at::ScalarType::Int)); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::zeros({4, 5}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[1].push_back(at::zeros({2, 3}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[2].push_back(at::zeros({2, 3}, {at::kCUDA}).to(at::ScalarType::Int)); - - add_sub_dynamic_test_helper(graph_IR, &addconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenSubIntdynamicConvertsCorrectly) { - // aten::sub.int(int a, int b) -> (int) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=0]() - %2 : int = prim::Constant[value=1]() - %3 : int = aten::size(%0, %1) - %4 : int = aten::size(%0, %2) - %5 : int = aten::sub(%3, %4) - %6 : Tensor = aten::add(%0, %5, %2) - return (%6))IR"; - baidu::mirana::poros::SubConverter subconverter; - std::vector input_data; - input_data.push_back(at::zeros({2, 3}, {at::kCUDA}).to(at::ScalarType::Int)); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::zeros({4, 5}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[1].push_back(at::zeros({2, 3}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[2].push_back(at::zeros({2, 3}, {at::kCUDA}).to(at::ScalarType::Int)); - - add_sub_dynamic_test_helper(graph_IR, &subconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenAddTdynamicConvertsCorrectly) { - // aten::add.t(t[] a, t[] b) -> (t[]) - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : int[] = aten::size(%0) - %3 : int[] = aten::size(%1) - %4 : int[] = aten::add(%2, %3) - %5 : int = prim::Constant[value=2]() - %6 : int = aten::__getitem__(%4, %5) - %7 : int = prim::Constant[value=1]() - %8 : Tensor = aten::add(%0, %6, %7) - return (%8))IR"; - baidu::mirana::poros::AddConverter addconverter; - std::vector input_data; - input_data.push_back(at::zeros({2, 3}, {at::kCUDA}).to(at::ScalarType::Int)); - input_data.push_back(at::zeros({4, 5}, {at::kCUDA}).to(at::ScalarType::Int)); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::zeros({4, 5}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[0].push_back(at::zeros({6, 7}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[1].push_back(at::zeros({2, 3}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[1].push_back(at::zeros({4, 5}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[2].push_back(at::zeros({2, 3}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[2].push_back(at::zeros({4, 5}, {at::kCUDA}).to(at::ScalarType::Int)); - - add_sub_dynamic_test_helper(graph_IR, &addconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenAddTensordynamicConvertsCorrectly) { - //dynamic tensor - const auto graph_IR = gen_add_sub_tensor_graph("add", "1.0"); - baidu::mirana::poros::AddConverter addconverter; - std::vector input_data; - input_data.push_back(at::randn({15, 1}, {at::kCUDA})); - input_data.push_back(at::randn({300}, {at::kCUDA})); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({40, 1}, {at::kCUDA})); - prewarm_data[0].push_back(at::randn({300}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({8, 1}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({300}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({20, 1}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({300}, {at::kCUDA})); - - add_sub_dynamic_test_helper(graph_IR, &addconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenAddTensordynamicMoreConvertsCorrectly) { - //dynamic tensor - const auto graph_IR = gen_add_sub_tensor_graph("add", "1.0"); - baidu::mirana::poros::AddConverter addconverter; - std::vector input_data; - input_data.push_back(at::randn({4, 1}, {at::kCUDA})); - input_data.push_back(at::randn({300}, {at::kCUDA})); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({4, 1}, {at::kCUDA})); - prewarm_data[0].push_back(at::randn({400}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({4, 1}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({100}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({4, 1}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({200}, {at::kCUDA})); - - add_sub_dynamic_test_helper(graph_IR, &addconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenAddTensordynamicMore2ConvertsCorrectly) { - //dynamic tensor - const auto graph_IR = gen_add_sub_tensor_graph("add", "1.0"); - baidu::mirana::poros::AddConverter addconverter; - std::vector input_data; - input_data.push_back(at::randn({4, 1, 45}, {at::kCUDA})); - input_data.push_back(at::randn({300, 1}, {at::kCUDA})); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({400, 1, 45}, {at::kCUDA})); - prewarm_data[0].push_back(at::randn({400, 1}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({4, 1, 45}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({100, 1}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({100, 1, 45}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({200, 1}, {at::kCUDA})); - - add_sub_dynamic_test_helper(graph_IR, &addconverter, input_data, true, &prewarm_data); -} \ No newline at end of file diff --git a/poros/unittest/converter/aten_eval_test.cpp b/poros/unittest/converter/aten_eval_test.cpp deleted file mode 100644 index 762906ee1f0..00000000000 --- a/poros/unittest/converter/aten_eval_test.cpp +++ /dev/null @@ -1,158 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file aten_eval_test.cpp -* @author wangrui39@baidu.com -* @date Mon December 13 11:36:11 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/aten_eval.h" -#include "poros/util/test_util.h" - -static void aten_eval_test_helper(const std::string& graph_IR, - const std::vector& input_data, - baidu::mirana::poros::IConverter* converter) { - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output)); - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - - -/*TEST(Converters, AppendConverterCorrectly) { - //"aten::append.t(t[](a!) self, t(c -> *) el) -> (t[](a!))" - - const auto graph = R"IR( - graph(%a.1 : Tensor, - %b.1 : Tensor, - %c.1 : Tensor): - %12 : int = prim::Constant[value=0]() - %x.1 : Tensor[] = prim::ListConstruct(%a.1) - %7 : Tensor[] = aten::append(%x.1, %b.1) # test.py:26:4 - %10 : Tensor[] = aten::append(%x.1, %c.1) # test.py:27:4 - %13 : Tensor = aten::cat(%x.1, %12) # test.py:28:11 - return (%13))IR"; - - std::vector input_data; - auto input1 = at::randn({3, 4}, {at::kCUDA}); - auto input2 = at::randn({3, 4}, {at::kCUDA}); - auto input3 = at::randn({3, 4}, {at::kCUDA}); - - input_data.push_back(input1); - input_data.push_back(input2); - input_data.push_back(input3); - - baidu::mirana::poros::AppendConverter appendConverter; - aten_eval_test_helper(graph, input_data, &appendConverter); -}*/ - -TEST(Converters, GetitemConverterCorrectly) { - // "aten::__getitem__.t(t[](a) list, int idx) -> (t(*))"*/ - - const auto graph = R"IR( - graph(%a.1 : Tensor, - %b.1 : Tensor): - %12 : int = prim::Constant[value=0]() - %16 : int = prim::Constant[value=1]() - %x.1 : Tensor[] = prim::ListConstruct(%a.1) - %7 : Tensor[] = aten::append(%x.1, %b.1) - %ret.1 : Tensor = aten::__getitem__(%x.1, %12) - %17 : Tensor = aten::__getitem__(%x.1, %16) - %19 : Tensor = aten::add(%ret.1, %17, %16) - return (%19))IR"; - - std::vector input_data; - auto input1 = at::randn({3, 4}, {at::kCUDA}); - auto input2 = at::randn({3, 4}, {at::kCUDA}); - - input_data.push_back(input1); - input_data.push_back(input2); - - baidu::mirana::poros::GetitemConverter getitemconverter; - aten_eval_test_helper(graph, input_data, &getitemconverter); -} - -TEST(Converters, SetitemConverterCorrectly) { - // aten::_set_item.t(t[](a!) l, int idx, t(b -> *) el) -> (t[](a!)) - const auto graph = R"IR( - graph(%x.1 : Tensor, - %y.1 : Tensor): - %6 : int = prim::Constant[value=1]() # test.py:28:15 - %10 : int = prim::Constant[value=0]() # test.py:28:6 - %a.1 : Tensor[] = prim::ListConstruct(%x.1, %y.1) - %8 : Tensor = aten::add(%x.1, %6, %6) # test.py:28:11 - %11 : Tensor[] = aten::_set_item(%a.1, %10, %8) # test.py:28:4 - %ret.1 : Tensor = aten::cat(%a.1, %10) # test.py:29:10 - return (%ret.1))IR"; - - std::vector input_data; - auto input1 = at::randn({3, 4}, {at::kCUDA}); - auto input2 = at::randn({3, 4}, {at::kCUDA}); - - input_data.push_back(input1); - input_data.push_back(input2); - - baidu::mirana::poros::SetitemConverter setitemconverter; - aten_eval_test_helper(graph, input_data, &setitemconverter); -} - -static void eval_dynamic_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - const std::vector& input_data, - bool is_dynamic = false, - std::vector>* prewarm_data = nullptr) { - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = is_dynamic; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output, prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenGetitemdynamicConvertsCorrectly) { - // "aten::__getitem__.t(t[](a) list, int idx) -> (t(*))"*/ - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int[] = aten::size(%0) - %2 : int = prim::Constant[value=1]() - %3 : int = aten::__getitem__(%1, %2) - %4 : Tensor = aten::add(%0, %3, %2) - return (%4))IR"; - baidu::mirana::poros::GetitemConverter getitemconverter; - std::vector input_data; - input_data.push_back(at::zeros({2, 3}, {at::kCUDA}).to(at::ScalarType::Int)); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::zeros({4, 5}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[1].push_back(at::zeros({2, 3}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[2].push_back(at::zeros({2, 3}, {at::kCUDA}).to(at::ScalarType::Int)); - - eval_dynamic_test_helper(graph_IR, &getitemconverter, input_data, true, &prewarm_data); -} diff --git a/poros/unittest/converter/batch_norm_test.cpp b/poros/unittest/converter/batch_norm_test.cpp deleted file mode 100644 index 08f5fbb7c28..00000000000 --- a/poros/unittest/converter/batch_norm_test.cpp +++ /dev/null @@ -1,146 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file batch_norm_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/batch_norm.h" -#include "poros/util/test_util.h" - -TEST(Converters, ATenBatchnormalConvertsCorrectly) { - // aten::batch_norm(Tensor input, Tensor? weight, Tensor? bias, Tensor? running_mean, Tensor? running_var, bool training, float momentum, float eps, bool cudnn_enabled) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1: Tensor, %2: Tensor, %3: Tensor, %4: Tensor): - %5 : bool = prim::Constant[value=0]() - %6 : float = prim::Constant[value=1.0000000000000001e-05]() - %7 : float = prim::Constant[value=0.10000000000000001]() - %8 : Tensor = aten::batch_norm(%0, %1, %2, %3, %4, %5, %6, %7, %5) - return (%8))IR"; - - auto in = at::randn({1, 5, 5, 5}, {at::kCUDA}); - auto gamma = at::randn({5}, {at::kCUDA}); - auto beta = at::randn({5}, {at::kCUDA}); - auto mean = at::randn({5}, {at::kCUDA}); - auto var = at::randn({5}, {at::kCUDA}).abs(); - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - baidu::mirana::poros::BatchNormConverter batchnormconverter; - - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &batchnormconverter, - {in, gamma, beta, mean, var}, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -/* -aten::instance_norm(Tensor input, -Tensor? weight, -Tensor? bias, -Tensor? running_mean, -Tensor? running_var, -bool use_input_stats, -float momentum, -float eps, -bool cudnn_enabled) -> Tensor -*/ -TEST(Converters, ATenInstanceNormConvertsCorrectly) { - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1: Tensor, %2: Tensor): - %3 : NoneType = prim::Constant() - %4 : bool = prim::Constant[value=1]() - %5 : float = prim::Constant[value=0.10000000000000001]() - %6 : float = prim::Constant[value=1.0000000000000001e-05]() - %7 : Tensor = aten::instance_norm(%0, %1, %2, %3, %3, %4, %5, %6, %4) - return (%7))IR"; - - auto input_tensor = at::randn({2, 10, 5, 5}, {at::kCUDA}); - auto weight = at::randn({10}, {at::kCUDA}); - auto bias = at::randn({10}, {at::kCUDA}); - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - baidu::mirana::poros::InstanceNormConverter instancenormconverter; - - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &instancenormconverter, - {input_tensor, weight, bias}, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenInstanceNormConvertsNoWeightCorrectly) { - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %3 : NoneType = prim::Constant() - %4 : bool = prim::Constant[value=1]() - %5 : float = prim::Constant[value=0.10000000000000001]() - %6 : float = prim::Constant[value=1.0000000000000001e-05]() - %7 : Tensor = aten::instance_norm(%0, %3, %3, %3, %3, %4, %5, %6, %4) - return (%7))IR"; - - auto input_tensor = at::randn({2, 20, 45, 3}, {at::kCUDA}); - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - baidu::mirana::poros::InstanceNormConverter instancenormconverter; - - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &instancenormconverter, - {input_tensor}, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenInstanceNormConverts3DCorrectly) { - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %3 : NoneType = prim::Constant() - %4 : bool = prim::Constant[value=1]() - %5 : float = prim::Constant[value=0.10000000000000001]() - %6 : float = prim::Constant[value=1.0000000000000001e-05]() - %7 : Tensor = aten::instance_norm(%0, %3, %3, %3, %3, %4, %5, %6, %4) - return (%7))IR"; - - auto input_tensor = at::randn({2, 20, 45}, {at::kCUDA}); - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - baidu::mirana::poros::InstanceNormConverter instancenormconverter; - - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &instancenormconverter, - {input_tensor}, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} \ No newline at end of file diff --git a/poros/unittest/converter/clone_test.cpp b/poros/unittest/converter/clone_test.cpp deleted file mode 100644 index f0153e2139b..00000000000 --- a/poros/unittest/converter/clone_test.cpp +++ /dev/null @@ -1,79 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file clone_test.cpp -* @author tianshaoqing@baidu.com -* @date Tue Nov 23 12:26:28 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/clone.h" -#include "poros/util/test_util.h" - -static void clone_dy_test_helper(const std::string& graph_IR, - const std::vector& input_data, - bool is_dynamic = false, - std::vector>* prewarm_data = nullptr) { - baidu::mirana::poros::CloneConverter cloneconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = is_dynamic; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &cloneconverter, - input_data, graph_output, poros_output, prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenCloneConvertsCorrectly) { - // aten::clone(Tensor self, *, MemoryFormat? memory_format=None) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %memory_format : None = prim::Constant[value=0]() - %1 : Tensor = aten::clone(%0, %memory_format) - %2 : Tensor = aten::relu(%1) - return (%2))IR"; - - std::vector input_data; - input_data.push_back(at::randn({10, 100, 100, 100}, {at::kCUDA})); - - clone_dy_test_helper(graph_IR, input_data); -} - -TEST(Converters, ATenCloneConvertsDynamicCorrectly) { - // aten::clone(Tensor self, *, MemoryFormat? memory_format=None) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %memory_format : None = prim::Constant[value=0]() - %1 : Tensor = aten::clone(%0, %memory_format) - %2 : Tensor = aten::relu(%1) - return (%2))IR"; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({20, 150, 100, 100}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({10, 100, 50, 50}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({10, 100, 50, 50}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({10, 100, 50, 50}, {at::kCUDA})); - - clone_dy_test_helper(graph_IR, input_data, true, &prewarm_data); -} \ No newline at end of file diff --git a/poros/unittest/converter/concat_test.cpp b/poros/unittest/converter/concat_test.cpp deleted file mode 100644 index b631fccc21b..00000000000 --- a/poros/unittest/converter/concat_test.cpp +++ /dev/null @@ -1,91 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file concat_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/concat.h" -#include "poros/util/test_util.h" - -static void cat_test_helper(const std::string& graph_IR, - std::vector shape1 = {5}, - std::vector shape2 = {5}, - bool Triple_inputs = false, - std::vector shape3 = {5}){ - std::vector input_data; - input_data.push_back(at::randn(shape1, {at::kCUDA})); - input_data.push_back(at::randn(shape2, {at::kCUDA})); - if (Triple_inputs){ - input_data.push_back(at::randn(shape3, {at::kCUDA})); - } - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - baidu::mirana::poros::ConcatConverter concatconverter; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &concatconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -static std::string gen_double_inputs_cat_graph(const std::string& dim) { - return R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : Tensor[] = prim::ListConstruct(%0, %1) - %3 : int = prim::Constant[value=)IR" + dim + R"IR(]() - %4 : Tensor = aten::cat(%2, %3) - return (%4))IR"; -} - -static std::string gen_triple_inputs_cat_graph(const std::string& dim) { - return R"IR( - graph(%0 : Tensor, %1 : Tensor, %2 : Tensor): - %3 : Tensor[] = prim::ListConstruct(%0, %1, %2) - %4 : int = prim::Constant[value=)IR" + dim + R"IR(]() - %5 : Tensor = aten::cat(%3, %4) - return (%5))IR"; -} - -TEST(Converters, ATenCatPureTensorConvertsCorrectly) { - // aten::cat(Tensor[] tensors, int dim=0) -> Tensor - const auto graph_IR = gen_double_inputs_cat_graph("0"); - cat_test_helper(graph_IR); -} - -TEST(Converters, ATenCatPureTensorNegDimConvertsCorrectly) { - // aten::cat(Tensor[] tensors, int dim=0) -> Tensor - const auto graph_IR = gen_double_inputs_cat_graph("-1"); - cat_test_helper(graph_IR, {5, 3}, {5, 4}); -} - -TEST(Converters, ATenCatTripleTensorConvertsCorrectly) { - // aten::cat(Tensor[] tensors, int dim=0) -> Tensor - const auto graph_IR = gen_triple_inputs_cat_graph("1"); - cat_test_helper(graph_IR, {5, 2, 2}, {5, 7, 2}, true, {5, 3, 2}); -} - -TEST(Converters, ATenCatTripleTensorNegdimConvertsCorrectly) { - // aten::cat(Tensor[] tensors, int dim=0) -> Tensor - const auto graph_IR = gen_triple_inputs_cat_graph("-1"); - cat_test_helper(graph_IR, {5, 6, 7}, {5, 6, 3}, true, {5, 6, 5}); -} \ No newline at end of file diff --git a/poros/unittest/converter/constant_pad_nd_test.cpp b/poros/unittest/converter/constant_pad_nd_test.cpp deleted file mode 100644 index f7aad85040c..00000000000 --- a/poros/unittest/converter/constant_pad_nd_test.cpp +++ /dev/null @@ -1,160 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file constant_pad_nd_test.cpp -* @author tianshaoqing@baidu.com -* @date Thur Dec 2 14:29:20 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/util/test_util.h" -#include "poros/converter/gpu/constant_pad_nd.h" - -static void constant_pad_nd_test_helper(const std::string& graph_IR, - std::vector input_data, - bool is_dynamic = false, - std::vector>* prewarm_data = nullptr) { - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = is_dynamic; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - baidu::mirana::poros::ConstantPadNdConverter constantpadndconverter; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &constantpadndconverter, - input_data, graph_output, poros_output, prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); - -} - -static std::string gen_constant_pad_nd_graph(const std::string& padding_shape_str, - const std::string& value_str, - const bool padding_value_is_int = false) { - if (padding_value_is_int) { - return R"IR( - graph(%0 : Tensor): - %1 : int[] = prim::Constant[value=[)IR" + padding_shape_str + R"IR(]]() - %2 : int = prim::Constant[value=)IR" + value_str + R"IR(]() - %3 : Tensor = aten::constant_pad_nd(%0, %1, %2) - return (%3))IR"; - - } else { - return R"IR( - graph(%0 : Tensor): - %1 : int[] = prim::Constant[value=[)IR" + padding_shape_str + R"IR(]]() - %2 : float = prim::Constant[value=)IR" + value_str + R"IR(]() - %3 : Tensor = aten::constant_pad_nd(%0, %1, %2) - return (%3))IR"; - } - -} - -TEST(Converters, TestAtenConstantPadNdCorrectly) { - const auto graph_IR = gen_constant_pad_nd_graph("1, 2, 3, 4", "1.5"); - std::vector input_data; - input_data.push_back(at::randn({4, 5, 6, 7}, {at::kCUDA})); - constant_pad_nd_test_helper(graph_IR, input_data); -} - -TEST(Converters, TestAtenConstantPadNdLastDimCorrectly) { - const auto graph_IR = gen_constant_pad_nd_graph("1, 2", "1.5"); - std::vector input_data; - input_data.push_back(at::randn({4, 5, 6, 7}, {at::kCUDA})); - constant_pad_nd_test_helper(graph_IR, input_data); -} - -TEST(Converters, TestAtenConstantPadNdZerosPaddingDimsCorrectly) { - const auto graph_IR = gen_constant_pad_nd_graph("0, 1, 2, 0", "1.5"); - std::vector input_data; - input_data.push_back(at::randn({4, 5, 6, 7}, {at::kCUDA})); - constant_pad_nd_test_helper(graph_IR, input_data); -} - -TEST(Converters, TestAtenConstantPadNdIntCorrectly) { - const auto graph_IR = gen_constant_pad_nd_graph("1, 2, 3, 4", "1", true); - std::vector input_data; - auto options_pyt = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kInt); - input_data.push_back(at::randint(0, 10, {4, 5, 6, 7}, options_pyt)); - constant_pad_nd_test_helper(graph_IR, input_data); -} - -TEST(Converters, TestAtenConstantPadNdInputSingleDimCorrectly) { - const auto graph_IR = gen_constant_pad_nd_graph("1, 2", "1.5"); - std::vector input_data; - input_data.push_back(at::randn({6}, {at::kCUDA})); - constant_pad_nd_test_helper(graph_IR, input_data); -} - -TEST(Converters, TestAtenConstantPadNdDynamicFloatCorrectly) { - const auto graph_IR = gen_constant_pad_nd_graph("1, 2, 3, 4", "1.5"); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({3, 4, 5, 6}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({2, 3, 4, 5}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({2, 3, 4, 5}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({2, 3, 4, 5}, {at::kCUDA})); - - constant_pad_nd_test_helper(graph_IR, input_data, true, &prewarm_data); -} - -TEST(Converters, TestAtenConstantPadNdDynamicFloatTwoPaddingDimsZerosCorrectly) { - const auto graph_IR = gen_constant_pad_nd_graph("2, 0, 0, 2", "1.5"); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({3, 4, 5, 6}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({2, 3, 4, 5}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({2, 3, 4, 5}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({2, 3, 4, 5}, {at::kCUDA})); - - constant_pad_nd_test_helper(graph_IR, input_data, true, &prewarm_data); -} - -TEST(Converters, TestAtenConstantPadNdDynamicFloatSingleDimCorrectly) { - const auto graph_IR = gen_constant_pad_nd_graph("1, 2", "1.5"); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({10}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({5}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({5}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({5}, {at::kCUDA})); - - constant_pad_nd_test_helper(graph_IR, input_data, true, &prewarm_data); -} - -TEST(Converters, TestAtenConstantPadNdDynamicIntCorrectly) { - const auto graph_IR = gen_constant_pad_nd_graph("1, 2, 3, 4", "2", true); - - std::vector> prewarm_data = {{}, {}, {}}; - auto options_pyt = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kFloat); - prewarm_data[0].push_back(at::randint(0, 10, {3, 4, 5, 6}, options_pyt)); - prewarm_data[1].push_back(at::randint(0, 10, {2, 3, 4, 5}, options_pyt)); - prewarm_data[2].push_back(at::randint(0, 10, {2, 3, 4, 5}, options_pyt)); - - std::vector input_data; - input_data.push_back(at::randint(0, 10, {2, 3, 4, 5}, {at::kCUDA})); - - constant_pad_nd_test_helper(graph_IR, input_data, true, &prewarm_data); -} \ No newline at end of file diff --git a/poros/unittest/converter/conv2d_test.cpp b/poros/unittest/converter/conv2d_test.cpp deleted file mode 100644 index bb38183c67d..00000000000 --- a/poros/unittest/converter/conv2d_test.cpp +++ /dev/null @@ -1,69 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file conv2d_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include -#include -#include - -#include "poros/util/test_util.h" -#include "poros/converter/gpu/convolution.h" - -static void conv2d_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - std::vector shape_inputs, - std::vector shape_weights, - std::vector shape_bias) { - std::vector input_data; - // auto in = at::randn({1, 3, 10, 10}, {at::kCUDA}); - // auto w = at::randn({8, 3, 5, 5}, {at::kCUDA}); - // auto b = at::randn({8}, {at::kCUDA}); - auto in = at::randn(shape_inputs, {at::kCUDA}); - auto w = at::randn(shape_weights, {at::kCUDA}); - auto b = at::randn(shape_bias, {at::kCUDA}); - input_data.push_back(in); - input_data.push_back(w); - input_data.push_back(b); - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - //ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 0.0001)); -} - -TEST(Converters, ATenConv2dVggishTestConvertsCorrectly) { - // aten::conv2d(Tensor input, Tensor weight, Tensor? bias=None, int[2] stride=1, int[2] padding=0, int[2] dilation=1, int groups=1) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor, %2 : Tensor): - %3 : int[] = prim::Constant[value=[1, 1]]() - %4 : int[] = prim::Constant[value=[1, 1]]() - %5 : int[] = prim::Constant[value=[1, 1]]() - %6 : int = prim::Constant[value=1]() - %7 : Tensor = aten::conv2d(%0, %1, %2, %3, %4, %5, %6) - return (%7))IR"; - baidu::mirana::poros::ConvolutionConverter convolutionconverter; - conv2d_test_helper(graph_IR, &convolutionconverter, {60, 256, 12, 8}, {512, 256, 3, 3}, {512}); -} diff --git a/poros/unittest/converter/einsum_test.cpp b/poros/unittest/converter/einsum_test.cpp deleted file mode 100644 index ca43249f9f4..00000000000 --- a/poros/unittest/converter/einsum_test.cpp +++ /dev/null @@ -1,150 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file einsum_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Jul 06 11:24:51 CST 2022 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/einsum.h" -#include "poros/util/test_util.h" - -static void aten_einsum_test_helper(const std::string& equation, - at::Tensor input1, - at::Tensor input2 = at::Tensor()) { - std::vector input_data; - input_data.push_back(input1); - if (input2.defined()) { - input_data.push_back(input2); - } - - std::string graph_IR; - if (input_data.size() == 2) { - graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %eq : str = prim::Constant[value=")IR" + equation + R"IR("]() - %2 : Tensor[] = prim::ListConstruct(%0, %1) - %3 : Tensor = aten::einsum(%eq, %2) - return (%3))IR"; - } else { - graph_IR = R"IR( - graph(%0 : Tensor): - %eq : str = prim::Constant[value=")IR" + equation + R"IR("]() - %2 : Tensor[] = prim::ListConstruct(%0) - %3 : Tensor = aten::einsum(%eq, %2) - return (%3))IR"; - } - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - baidu::mirana::poros::EinsumConverter einsumconverter; - - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &einsumconverter, - input_data, graph_output, poros_output)); - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenEinsumConverterCorrectly) { -// aten::einsum(str equation, Tensor[] tensors) -> (Tensor) - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %eq : str = prim::Constant[value="bfnd,ndh->bfh"]() - %2 : Tensor[] = prim::ListConstruct(%0, %1) - %3 : Tensor = aten::einsum(%eq, %2) - return (%3))IR"; - - std::vector input_data; - - auto options_pyt_float = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kFloat); - input_data.push_back(at::randn({20, 30, 12, 26}, options_pyt_float)); - input_data.push_back(at::randn({12, 26, 312}, options_pyt_float)); - - baidu::mirana::poros::EinsumConverter einsumconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = false; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &einsumconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenEinsumTorchExamplesTestConverterCorrectly) { - // Test cases from https://gist.github.com/rockt/15ee013889d65342088e9260a377dc8f - auto options_pyt_float = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kFloat); - at::Tensor x = at::randn({5}, options_pyt_float); - at::Tensor y = at::randn({7}, options_pyt_float); - at::Tensor A = at::randn({3, 5}, options_pyt_float); - at::Tensor B = at::randn({2, 5}, options_pyt_float); - at::Tensor C = at::randn({2, 3, 5}, options_pyt_float); - at::Tensor D = at::randn({2, 5, 7}, options_pyt_float); - at::Tensor E = at::randn({7, 9}, options_pyt_float); - at::Tensor F = at::randn({2, 3, 3, 5}, options_pyt_float); - at::Tensor G = at::randn({5, 4, 6}, options_pyt_float); - at::Tensor H = at::randn({4, 4}, options_pyt_float); - at::Tensor I = at::randn({2, 3, 2}, options_pyt_float); - - // vector operations - aten_einsum_test_helper("i->", x); // sum - aten_einsum_test_helper("i,i->", x, x); // dot - aten_einsum_test_helper("i,i->i", x, x); // vector element-wisem mul - aten_einsum_test_helper("i,j->j", x, y); // outer - - // Matrix operations - aten_einsum_test_helper("ij->ji", A); // transpose - aten_einsum_test_helper("ij->j", A); // row sum - aten_einsum_test_helper("ij->i", A); // col sum - aten_einsum_test_helper("ij,ij->ij", A, A); // matrix element-wise mul - aten_einsum_test_helper("ij,j->i", A, x); // matrix vector multiplication - aten_einsum_test_helper("ij,kj->ik", A, B); // matmul - aten_einsum_test_helper("ij,ab->ijab", A, E); // matrix outer product - - // Tensor operations - aten_einsum_test_helper("Aij,Ajk->Aik", C, D); // batch matmul - aten_einsum_test_helper("ijk,jk->i", C, A); // tensor matrix contraction - aten_einsum_test_helper("aij,jk->aik", D, E); // tensor matrix contraction - aten_einsum_test_helper("abCd,dfg->abCfg", F, G); // tensor tensor contraction - aten_einsum_test_helper("ijk,jk->ik", C, A); // tensor matrix contraction with double indices - aten_einsum_test_helper("ijk,jk->ij", C, A); // tensor matrix contraction with double indices - aten_einsum_test_helper("ijk,ik->j", C, B); // non contiguous - aten_einsum_test_helper("ijk,ik->jk", C, B); // non contiguous with double indices - - // Diagonal operations are not permitted in poros - // aten_einsum_test_helper("ii", H); // trace - // aten_einsum_test_helper("ii->i", H); // diagonal - // aten_einsum_test_helper("iji->j", I); // non-contiguous trace - // aten_einsum_test_helper("ngrg...->nrg...", at::randn({2, 1, 3, 1, 4}, options_pyt_float)); - - // Ellipsis equations are not permitted in poros - // aten_einsum_test_helper("i...->...", H); - // aten_einsum_test_helper("ki,...k->i...", A.t(), B); - // aten_einsum_test_helper("k...,jk->...", A.t(), B); - // aten_einsum_test_helper('...ik, ...j -> ...ij', C, x); - // aten_einsum_test_helper('Bik,k...j->i...j', C, at::randn({5, 3}, options_pyt_float)); - // aten_einsum_test_helper('i...j, ij... -> ...ij', C, at::randn({2, 5, 2, 3}, options_pyt_float)); -} \ No newline at end of file diff --git a/poros/unittest/converter/element_wise_test.cpp b/poros/unittest/converter/element_wise_test.cpp deleted file mode 100644 index 80d23b4a272..00000000000 --- a/poros/unittest/converter/element_wise_test.cpp +++ /dev/null @@ -1,305 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file element_wise_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/element_wise.h" -#include "poros/util/test_util.h" - -static void poros_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - const std::vector& input_data){ - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - std::string pow_node_name("aten::pow"); - if(converter->node_kind()[0].toQualString() == pow_node_name){ - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); - }else{ - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); - } -} - -static void pow_test_examples(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - bool singleInput, - std::vector shape1 = {5}, - std::vector shape2 = {5}){ - std::vector input_data; - input_data.push_back(at::randn(shape1, {at::kCUDA})); - if (!singleInput){ - input_data.push_back(at::randint(-5, 5, shape2, {at::kCUDA})); - } - poros_test_helper(graph_IR, converter, input_data); -} - -TEST(Converters, ATenPowTensorConvertsCorrectly) { - // aten::pow.Tensor_Tensor(Tensor self, Tensor exponent) -> Tensor - const auto graph_IR = R"IR( - graph(%1 : Tensor, %2 : Tensor): - %3 : Tensor = aten::pow(%1, %2) - return (%3))IR"; - baidu::mirana::poros::PowOrFloordivideConverter poworfloordivideconverter; - pow_test_examples(graph_IR, &poworfloordivideconverter, false); - pow_test_examples(graph_IR, &poworfloordivideconverter, false, {3, 4}, {4}); - pow_test_examples(graph_IR, &poworfloordivideconverter, false, {4}, {3, 4}); - pow_test_examples(graph_IR, &poworfloordivideconverter, false, {3, 4, 3}, {4, 3}); - pow_test_examples(graph_IR, &poworfloordivideconverter, false, {4, 3}, {3, 4, 3}); -} - -TEST(Converters, ATenPowScalarConvertsCorrectly) { - // aten::pow.Tensor_Scalar(Tensor self, Scalar exponent) -> Tensor - const auto graph_IR = R"IR( - graph(%1 : Tensor): - %2 : float = prim::Constant[value=2.0]() - %3 : Tensor = aten::pow(%1, %2) - return (%3))IR"; - baidu::mirana::poros::PowOrFloordivideConverter poworfloordivideconverter; - pow_test_examples(graph_IR, &poworfloordivideconverter, true); - pow_test_examples(graph_IR, &poworfloordivideconverter, true, {3, 4}); -} - -static void elementwise_tensor_test_examples(const std::string& op, - baidu::mirana::poros::IConverter* converter){ - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : Tensor = aten::)IR" + op + R"IR((%0, %1) - return (%2))IR"; - std::vector input_data; - input_data.push_back(at::randn({2, 2}, {at::kCUDA})); - input_data.push_back(at::randn({2, 2}, {at::kCUDA})); - poros_test_helper(graph_IR, converter, input_data); - - input_data.clear(); - input_data.push_back(at::randn({2, 2}, {at::kCUDA})); - input_data.push_back(at::randn({2, 2}, {at::kCUDA})); - input_data[0][0][0] = 2.5; - input_data[1][0][0] = 2.5; - poros_test_helper(graph_IR, converter, input_data); - - input_data.clear(); - input_data.push_back(at::randn({3, 4, 3}, {at::kCUDA})); - input_data.push_back(at::randn({4, 3}, {at::kCUDA})); - input_data[0][0][0][0] = 2.5; - input_data[1][0][0] = 2.5; - poros_test_helper(graph_IR, converter, input_data); - - input_data.clear(); - input_data.push_back(at::randn({4, 3}, {at::kCUDA})); - input_data.push_back(at::randn({3, 4, 3}, {at::kCUDA})); - input_data[0][0][0] = 2.5; - input_data[1][0][0][0] = 2.5; - poros_test_helper(graph_IR, converter, input_data); -} - -static void elementwise_scalar_test_examples(const std::string& op, - const std::string& scalar, - baidu::mirana::poros::IConverter* converter){ - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : float = prim::Constant[value=)IR" + scalar + R"IR(]() - %2 : Tensor = aten::)IR" + op + R"IR((%0, %1) - return (%2))IR"; - std::vector input_data; - input_data.push_back(at::randn({2, 2}, {at::kCUDA})); - poros_test_helper(graph_IR, converter, input_data); - - input_data.clear(); - input_data.push_back(at::randn({2, 2}, {at::kCUDA})); - input_data[0][0][0] = 2.5; - poros_test_helper(graph_IR, converter, input_data); - - input_data.clear(); - input_data.push_back(at::randn({1}, {at::kCUDA})); - input_data[0][0] = 2.5; - poros_test_helper(graph_IR, converter, input_data); -} - -TEST(Converters, ATenEqualTensorConvertsCorrectly) { - // aten::eq.Tensor(Tensor self, Tensor other) -> Tensor - baidu::mirana::poros::EqualOrNotequalConverter equalorbotequalconverter; - elementwise_tensor_test_examples("eq", &equalorbotequalconverter); -} - -TEST(Converters, ATenEqualScalarConvertsCorrectly) { - // aten::eq.Scalar(Tensor self, Scalar other) -> Tensor - baidu::mirana::poros::EqualOrNotequalConverter equalorbotequalconverter; - elementwise_scalar_test_examples("eq", "2.5",&equalorbotequalconverter); -} - -TEST(Converters, ATenNotEqualTensorConvertsCorrectly) { - // aten::ne.Tensor(Tensor self, Tensor other) -> Tensor - baidu::mirana::poros::EqualOrNotequalConverter equalorbotequalconverter; - elementwise_tensor_test_examples("ne", &equalorbotequalconverter); -} - -TEST(Converters, ATenNotEqualScalarConvertsCorrectly) { - // aten::ne.Scalar(Tensor self, Scalar other) -> Tensor - baidu::mirana::poros::EqualOrNotequalConverter equalorbotequalconverter; - elementwise_scalar_test_examples("ne", "2.5", &equalorbotequalconverter); -} - -TEST(Converters, ATenGtTensorConvertsCorrectly) { - // aten::gt.Tensor(Tensor self, Tensor other) -> Tensor - baidu::mirana::poros::GreaterOrLessConverter greaterorlessconverter; - elementwise_tensor_test_examples("gt", &greaterorlessconverter); -} - -TEST(Converters, ATenGtScalarConvertsCorrectly) { - // aten::gt.Scalar(Tensor self, Scalar other) -> Tensor - baidu::mirana::poros::GreaterOrLessConverter greaterorlessconverter; - elementwise_scalar_test_examples("gt", "2.5", &greaterorlessconverter); -} - -TEST(Converters, ATenLtTensorConvertsCorrectly) { - // aten::lt.Tensor(Tensor self, Tensor other) -> Tensor - baidu::mirana::poros::GreaterOrLessConverter greaterorlessconverter; - elementwise_tensor_test_examples("lt", &greaterorlessconverter); -} - -TEST(Converters, ATenLtScalarConvertsCorrectly) { - // aten::lt.Scalar(Tensor self, Scalar other) -> Tensor - baidu::mirana::poros::GreaterOrLessConverter greaterorlessconverter; - elementwise_scalar_test_examples("lt", "2.5", &greaterorlessconverter); -} - -TEST(Converters, ATenGeTensorConvertsCorrectly) { - // aten::ge.Tensor(Tensor self, Tensor other) -> Tensor - baidu::mirana::poros::GreaterOrLessConverter greaterorlessconverter; - elementwise_tensor_test_examples("ge", &greaterorlessconverter); -} - -TEST(Converters, ATenGeScalarConvertsCorrectly) { - // aten::ge.Scalar(Tensor self, Scalar other) -> Tensor - baidu::mirana::poros::GreaterOrLessConverter greaterorlessconverter; - elementwise_scalar_test_examples("ge", "2.5", &greaterorlessconverter); -} - -TEST(Converters, ATenLeTensorConvertsCorrectly) { - // aten::le.Tensor(Tensor self, Tensor other) -> Tensor - baidu::mirana::poros::GreaterOrLessConverter greaterorlessconverter; - elementwise_tensor_test_examples("le", &greaterorlessconverter); -} - -TEST(Converters, ATenLeScalarConvertsCorrectly) { - // aten::le.Scalar(Tensor self, Scalar other) -> Tensor - baidu::mirana::poros::GreaterOrLessConverter greaterorlessconverter; - elementwise_scalar_test_examples("le", "2.5", &greaterorlessconverter); -} - -static std::string gen_clamp_graph(const std::string& op, - const std::string& min_val, - const std::string& max_val){ - if (op == "clamp"){ - std::string min_val_IR; - std::string max_val_IR; - if (min_val.empty()){ - min_val_IR = "None = prim::Constant()"; - }else{ - min_val_IR = "float = prim::Constant[value=" + min_val + "]()"; - } - if (max_val.empty()){ - max_val_IR = "None = prim::Constant()"; - }else{ - max_val_IR = "float = prim::Constant[value=" + max_val + "]()"; - } - return R"IR( - graph(%0 : Tensor): - %1 : )IR" + min_val_IR + R"IR( - %2 : )IR" + max_val_IR + R"IR( - %3 : Tensor = aten::)IR" + op + R"IR((%0, %1, %2) - return (%3))IR"; - }else if (op == "clamp_min"){ - return R"IR( - graph(%0 : Tensor): - %1 : float = prim::Constant[value=)IR" + min_val + R"IR(]() - %2 : Tensor = aten::)IR" + op + R"IR((%0, %1) - return (%2))IR"; - }else if (op == "clamp_max"){ - return R"IR( - graph(%0 : Tensor): - %1 : float = prim::Constant[value=)IR" + max_val + R"IR(]() - %2 : Tensor = aten::)IR" + op + R"IR((%0, %1) - return (%2))IR"; - }else{ - return ""; - } -} - -TEST(Converters, ATenClampMinConvertsCorrectly) { - // aten::clamp(Tensor self, Scalar? min=None, Scalar? max=None) -> Tensor - const auto graph_IR = gen_clamp_graph("clamp", "1.5", ""); - baidu::mirana::poros::ClampConverter clampconverter; - std::vector input_data; - input_data.push_back(at::randn({10}, {at::kCUDA})); - poros_test_helper(graph_IR, &clampconverter, input_data); -} - -TEST(Converters, ATenClampMaxConvertsCorrectly) { - // aten::clamp(Tensor self, Scalar? min=None, Scalar? max=None) -> Tensor - const auto graph_IR = gen_clamp_graph("clamp", "", "0.5"); - baidu::mirana::poros::ClampConverter clampconverter; - std::vector input_data; - input_data.push_back(at::randn({10}, {at::kCUDA})); - poros_test_helper(graph_IR, &clampconverter, input_data); -} - -TEST(Converters, ATenClampMinMaxConvertsCorrectly) { - // aten::clamp(Tensor self, Scalar? min=None, Scalar? max=None) -> Tensor - const auto graph_IR = gen_clamp_graph("clamp", "-0.5", "0.5"); - baidu::mirana::poros::ClampConverter clampconverter; - std::vector input_data; - input_data.push_back(at::randn({10}, {at::kCUDA})); - poros_test_helper(graph_IR, &clampconverter, input_data); -} - -TEST(Converters, ATenClampMaximumConvertsCorrectly) { - // aten::clamp_max(Tensor self, Scalar max) -> Tensor - const auto graph_IR = gen_clamp_graph("clamp_max", "", "0.5"); - baidu::mirana::poros::ClampConverter clampconverter; - std::vector input_data; - input_data.push_back(at::randn({10}, {at::kCUDA})); - poros_test_helper(graph_IR, &clampconverter, input_data); -} - -TEST(Converters, ATenClampMinimumConvertsCorrectly) { - // aten::clamp_min(Tensor self, Scalar min) -> Tensor - const auto graph_IR = gen_clamp_graph("clamp_min", "-0.5", ""); - baidu::mirana::poros::ClampConverter clampconverter; - std::vector input_data; - input_data.push_back(at::randn({10}, {at::kCUDA})); - poros_test_helper(graph_IR, &clampconverter, input_data); -} - -TEST(Converters, ATenClampMinGtMaxConvertsCorrectly) { - // aten::clamp_min(Tensor self, Scalar min) -> Tensor - const auto graph_IR = gen_clamp_graph("clamp", "0.5", "-0.5"); - baidu::mirana::poros::ClampConverter clampconverter; - std::vector input_data; - input_data.push_back(at::randn({10}, {at::kCUDA})); - poros_test_helper(graph_IR, &clampconverter, input_data); -} \ No newline at end of file diff --git a/poros/unittest/converter/expand_test.cpp b/poros/unittest/converter/expand_test.cpp deleted file mode 100644 index e2221cef9ba..00000000000 --- a/poros/unittest/converter/expand_test.cpp +++ /dev/null @@ -1,233 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file expand_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/expand.h" -#include "poros/util/test_util.h" - -static void expand_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - bool singleInput, - std::vector shape1 = {3, 1}, - std::vector shape2 = {3, 1}){ - std::vector input_data; - input_data.push_back(at::randn(shape1, {at::kCUDA})); - if (!singleInput){ - input_data.push_back(at::randn(shape2, {at::kCUDA})); - } - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - // ASSERT_TRUE(baidu::mirana::poros::testutil::almostEqual(graph_output[0], poros_output[0], 2e-6)); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -static std::string gen_expand_graph(const std::string& size, const std::string& implicit) { - return R"IR( - graph(%0 : Tensor): - %1 : int[] = prim::Constant[value=[)IR" + size + R"IR(]]() - %2 : bool = prim::Constant[value=)IR" + implicit + R"IR(]() - %3 : Tensor = aten::expand(%0, %1, %2) - return (%3))IR"; -} - -static std::string gen_repeat_graph(const std::string& size) { - return R"IR( - graph(%0 : Tensor): - %1 : int[] = prim::Constant[value=[)IR" + size + R"IR(]]() - %2 : Tensor = aten::repeat(%0, %1) - return (%2))IR"; -} - -TEST(Converters, ATenExpandSameDimConvertsCorrectly) { - // aten::expand(Tensor(a) self, int[] size, *, bool implicit=False) -> Tensor(a) - const auto graph_IR = gen_expand_graph("3, 4", "0"); - baidu::mirana::poros::ExpandConverter expandconverter; - expand_test_helper(graph_IR, &expandconverter, true); -} - -TEST(Converters, ATenExpandTileConvertsCorrectly) { - // aten::expand(Tensor(a) self, int[] size, *, bool implicit=False) -> Tensor(a) - // 若%2参数个数大于%1,则expand从后向前对齐 - // [3,1] [2,3,4] -> [2,3,4] - // [3,1] [1,3,4] -> [1,3,4] - // [3,1] [3,-1,4] -> [3,3,4] - const auto graph_IR = gen_expand_graph("2, 3, 4", "0"); - baidu::mirana::poros::ExpandConverter expandconverter; - expand_test_helper(graph_IR, &expandconverter, true); -} - -TEST(Converters, ATenExpandTileLastConvertsCorrectly) { - // aten::expand(Tensor(a) self, int[] size, *, bool implicit=False) -> Tensor(a) - const auto graph_IR = gen_expand_graph("1, 3, 4", "0"); - baidu::mirana::poros::ExpandConverter expandconverter; - expand_test_helper(graph_IR, &expandconverter, true); -} - -TEST(Converters, ATenExpandNegativeSizeConvertsCorrectly) { - // aten::expand(Tensor(a) self, int[] size, *, bool implicit=False) -> Tensor(a) - // 1 means not changing the size of that dimension - const auto graph_IR = gen_expand_graph("3, -1, 4", "0"); - baidu::mirana::poros::ExpandConverter expandconverter; - expand_test_helper(graph_IR, &expandconverter, true); -} - -TEST(Converters, ATenRepeatConvertsCorrectly) { - // aten::repeat(Tensor self, int[] repeats) -> Tensor - // output shape计算方法:参数向后对齐(如果%1与%2维度不同的话,同expand),依次相乘 - // [3,1] [4,2] -> [12,2] - // [2,3,2] [2,2,2] -> [4,6,4] - // [3,1] [1,3,2] -> [1,9,2] - const auto graph_IR = gen_repeat_graph("4, 2"); - baidu::mirana::poros::RepeatConverter repeatconverter; - expand_test_helper(graph_IR, &repeatconverter, true); -} - -TEST(Converters, ATenRepeat3dConvertsCorrectly) { - // aten::repeat(Tensor self, int[] repeats) -> Tensor - const auto graph_IR = gen_repeat_graph("2, 2, 2"); - baidu::mirana::poros::RepeatConverter repeatconverter; - expand_test_helper(graph_IR, &repeatconverter, true, {2, 3, 2}); -} - -TEST(Converters, ATenRepeatExtraDimsConvertsCorrectly) { - // aten::repeat(Tensor self, int[] repeats) -> Tensor - const auto graph_IR = gen_repeat_graph("1, 3, 2"); - baidu::mirana::poros::RepeatConverter repeatconverter; - expand_test_helper(graph_IR, &repeatconverter, true); -} - -static void expand_dynamic_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - const std::vector& input_data, - bool is_dynamic = false, - std::vector>* prewarm_data = nullptr) { - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = is_dynamic; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output, prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenExpandFromSizedynamicConvertsCorrectly) { - // aten::expand(Tensor(a) self, int[] size, *, bool implicit=False) -> Tensor(a) - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : int = prim::Constant[value=-1]() - %3 : int[] = aten::size(%0) - %B.1 : int, %H.1 : int, %W.1 : int, %C.1 : int = prim::ListUnpack(%3) - %4 : int[] = prim::ListConstruct(%B.1, %2, %C.1) - %5 : Tensor = aten::reshape(%0, %4) - %6 : int[] = aten::size(%5) - %B.2 : int, %N.2 : int, %C.2 : int = prim::ListUnpack(%6) - %7 : int[] = prim::ListConstruct(%B.2, %2, %2) - %8 : bool = prim::Constant[value=0]() - %9 : Tensor = aten::expand(%1, %7, %8) - return (%9))IR"; - baidu::mirana::poros::ExpandConverter expandconverter; - std::vector input_data; - input_data.push_back(at::randn({2, 24, 24, 512}, {at::kCUDA})); - input_data.push_back(at::randn({1, 1, 512}, {at::kCUDA})); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({4, 24, 24, 512}, {at::kCUDA})); - prewarm_data[0].push_back(at::randn({1, 1, 512}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({2, 24, 24, 512}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({1, 1, 512}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({2, 24, 24, 512}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({1, 1, 512}, {at::kCUDA})); - - expand_dynamic_test_helper(graph_IR, &expandconverter, input_data, true, &prewarm_data); -} - - -/*aten::expand_as(Tensor(a) self, Tensor other) -> Tensor(a)*/ -static std::string gen_expand_as_graph() { - return R"IR( - graph(%0 : Tensor, %1 : Tensor): - %3 : Tensor = aten::expand_as(%0, %1) - return (%3))IR"; -} - -TEST(Converters, ATenExpandAsConvertsCorrectly) { - /*aten::expand_as(Tensor(a) self, Tensor other) -> Tensor(a)*/ - const auto graph_IR = gen_expand_as_graph(); - baidu::mirana::poros::ExpandConverter expandconverter; - - std::vector input_data; - input_data.push_back(at::randn({1, 1, 512}, {at::kCUDA})); - input_data.push_back(at::randn({2, 24, 1, 512}, {at::kCUDA})); - - expand_dynamic_test_helper(graph_IR, &expandconverter, input_data); -} - -TEST(Converters, ATenExpandAsDynamicConvertsCorrectly) { - /*aten::expand_as(Tensor(a) self, Tensor other) -> Tensor(a)*/ - const auto graph_IR = gen_expand_as_graph(); - baidu::mirana::poros::ExpandConverter expandconverter; - - std::vector input_data; - input_data.push_back(at::randn({1, 1, 512}, {at::kCUDA})); - input_data.push_back(at::randn({2, 24, 1, 512}, {at::kCUDA})); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({1, 1, 512}, {at::kCUDA})); - prewarm_data[0].push_back(at::randn({4, 24, 1, 512}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({1, 1, 512}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({2, 24, 1, 512}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({1, 1, 512}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({2, 24, 1, 512}, {at::kCUDA})); - - expand_dynamic_test_helper(graph_IR, &expandconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenExpandAsDynamicMoreConvertsCorrectly) { - /*aten::expand_as(Tensor(a) self, Tensor other) -> Tensor(a)*/ - const auto graph_IR = gen_expand_as_graph(); - baidu::mirana::poros::ExpandConverter expandconverter; - - std::vector input_data; - input_data.push_back(at::randn({24, 1, 512}, {at::kCUDA})); - input_data.push_back(at::randn({4, 24, 1, 512}, {at::kCUDA})); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({24, 1, 512}, {at::kCUDA})); - prewarm_data[0].push_back(at::randn({4, 24, 1, 512}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({2, 1, 512}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({4, 2, 1, 512}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({4, 1, 512}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({4, 4, 1, 512}, {at::kCUDA})); - - expand_dynamic_test_helper(graph_IR, &expandconverter, input_data, true, &prewarm_data); -} \ No newline at end of file diff --git a/poros/unittest/converter/generate_test.cpp b/poros/unittest/converter/generate_test.cpp deleted file mode 100644 index 261d893075c..00000000000 --- a/poros/unittest/converter/generate_test.cpp +++ /dev/null @@ -1,546 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file generate_test.cpp -* @author tianshaoqing@baidu.com -* @date Tue Nov 23 12:26:28 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/generate.h" -#include "poros/util/test_util.h" - -static void generate_dy_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - const std::vector& input_data, - bool is_dynamic = false, - std::vector>* prewarm_data = nullptr) { - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = is_dynamic; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output, prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenZeroslikeConvertsCorrectly) { - // aten::zeros_like(Tensor self, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None, MemoryFormat? memory_format=None) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : None = prim::Constant() - %zerosout : Tensor = aten::zeros_like(%0, %1, %1, %1, %1, %1) - return (%zerosout))IR"; - std::vector input_data; - input_data.push_back(at::randn({5, 6, 7}, {at::kCUDA})); - baidu::mirana::poros::ZerosLikeConverter zeroslikeconverter; - generate_dy_test_helper(graph_IR, &zeroslikeconverter, input_data); -} - -TEST(Converters, ATenZeroslikeDtypeConvertsCorrectly) { - // aten::zeros_like(Tensor self, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None, MemoryFormat? memory_format=None) -> Tensor - // scalar type index in aten and support situation ('o' is support and 'x' is not support): - // uint8_t -> 0 x - // int8_t -> 1 x - // int16_t -> 2 x - // int -> 3 o - // int64_t -> 4 x - // Half -> 5 o - // float -> 6 o - // bool -> 11 x - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : None = prim::Constant() - %2 : int = prim::Constant[value=3]() - %zerosout : Tensor = aten::zeros_like(%0, %2, %1, %1, %1, %1) - return (%zerosout))IR"; - std::vector input_data; - input_data.push_back(at::randn({5, 6, 7}, {at::kCUDA})); - baidu::mirana::poros::ZerosLikeConverter zeroslikeconverter; - generate_dy_test_helper(graph_IR, &zeroslikeconverter, input_data); -} - -TEST(Converters, ATenZeroslikeDynamicConvertsCorrectly) { - // aten::zeros_like(Tensor self, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None, MemoryFormat? memory_format=None) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : None = prim::Constant() - %zerosout : Tensor = aten::zeros_like(%0, %1, %1, %1, %1, %1) - return (%zerosout))IR"; - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({5, 6, 7}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({4, 5, 6}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({4, 5, 6}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({4, 5, 6}, {at::kCUDA})); - baidu::mirana::poros::ZerosLikeConverter zeroslikeconverter; - generate_dy_test_helper(graph_IR, &zeroslikeconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenZeroslikeDynamicDtypeConvertsCorrectly) { - // aten::zeros_like(Tensor self, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None, MemoryFormat? memory_format=None) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : None = prim::Constant() - %2 : int = prim::Constant[value=5]() - %zerosout : Tensor = aten::zeros_like(%0, %2, %1, %1, %1, %1) - return (%zerosout))IR"; - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({5, 6, 7}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({4, 5, 6}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({4, 5, 6}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({4, 5, 6}, {at::kCUDA})); - baidu::mirana::poros::ZerosLikeConverter zeroslikeconverter; - generate_dy_test_helper(graph_IR, &zeroslikeconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenZerosDynamicConvertsCorrectly) { - // aten::zeros(int[] size, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int[] = aten::size(%0) - %2 : None = prim::Constant() - %3 : Device = prim::Constant[value="cuda"]() - %4 : Tensor = aten::zeros(%1, %2, %2, %3, %2) - return (%4))IR"; - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({5, 6, 7}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({4, 5, 6}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({4, 5, 6}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({4, 5, 6}, {at::kCUDA})); - baidu::mirana::poros::ZerosConverter ZerosConverter; - generate_dy_test_helper(graph_IR, &ZerosConverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenZerosDynamicDtypeConvertsCorrectly) { - // aten::zeros(int[] size, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int[] = aten::size(%0) - %2 : None = prim::Constant() - %3 : Device = prim::Constant[value="cuda"]() - %4 : int = prim::Constant[value=3]() - %5 : Tensor = aten::zeros(%1, %4, %2, %3, %2) - return (%5))IR"; - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({5, 6, 7}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({4, 5, 6}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({4, 5, 6}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({4, 5, 6}, {at::kCUDA})); - baidu::mirana::poros::ZerosConverter ZerosConverter; - generate_dy_test_helper(graph_IR, &ZerosConverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenOnesDynamicConvertsCorrectly) { - // aten::ones(int[] size, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int[] = aten::size(%0) - %2 : None = prim::Constant() - %3 : Device = prim::Constant[value="cuda"]() - %4 : Tensor = aten::ones(%1, %2, %2, %3, %2) - return (%4))IR"; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({5, 6, 7}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({4, 5, 6}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({4, 5, 6}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({4, 5, 6}, {at::kCUDA})); - baidu::mirana::poros::OnesConverter onesconverter; - generate_dy_test_helper(graph_IR, &onesconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenOnesDynamicDtypeConvertsCorrectly) { - // aten::ones(int[] size, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int[] = aten::size(%0) - %2 : None = prim::Constant() - %3 : Device = prim::Constant[value="cuda"]() - %4 : int = prim::Constant[value=5]() - %5 : Tensor = aten::ones(%1, %4, %2, %3, %2) - return (%5))IR"; - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({5, 6, 7}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({4, 5, 6}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({4, 5, 6}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({4, 5, 6}, {at::kCUDA})); - baidu::mirana::poros::OnesConverter onesconverter; - generate_dy_test_helper(graph_IR, &onesconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenFullDynamicDtypeConvertsCorrectly) { - // aten::full(int[] size, Scalar fill_value, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int[] = aten::size(%0) - %2 : None = prim::Constant() - %3 : Device = prim::Constant[value="cuda"]() - %4 : int = prim::Constant[value=6]() - %5 : Tensor = aten::full(%1, %4, %4, %2, %3, %2) - return (%5))IR"; - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({5, 6, 7}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({4, 5, 6}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({4, 5, 6}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({4, 5, 6}, {at::kCUDA})); - baidu::mirana::poros::FullConverter fullconverter; - generate_dy_test_helper(graph_IR, &fullconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenArangeDynamicDtypeConvertsCorrectly) { - // aten::arange(Scalar end, *, ScalarType? dtype=None, Layout? layout=None, Device? device=None, bool? pin_memory=None) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=1]() - %2 : int = aten::size(%0, %1) - %3 : None = prim::Constant() - %4 : Device = prim::Constant[value="cuda"]() - %5 : int = prim::Constant[value=3]() - %6 : Tensor = aten::arange(%2, %5, %3, %4, %3) - return (%6))IR"; - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({5, 6, 7}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({4, 5, 6}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({4, 5, 6}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({4, 5, 6}, {at::kCUDA})); - baidu::mirana::poros::ArangeConverter arangeconverter; - generate_dy_test_helper(graph_IR, &arangeconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenArangeStartEndDynamicDtypeConvertsCorrectly) { - // aten::arange.start(Scalar start, Scalar end, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None) -> (Tensor) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=0]() - %2 : int = prim::Constant[value=1]() - %s.1 : int = aten::size(%0, %1) - %s.2 : int = aten::size(%0, %2) - %3 : None = prim::Constant() - %4 : Device = prim::Constant[value="cuda"]() - %5 : int = prim::Constant[value=3]() - %6 : Tensor = aten::arange(%s.1, %s.2, %5, %3, %4, %3) - return (%6))IR"; - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({1, 8}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({1, 2}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({1, 5}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({1, 5}, {at::kCUDA})); - baidu::mirana::poros::ArangeConverter arangeconverter; - generate_dy_test_helper(graph_IR, &arangeconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenArangeStartConstantEndDynamicDtypeConvertsCorrectly) { - // aten::arange.start(Scalar start, Scalar end, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None) -> (Tensor) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %s.1 : int = prim::Constant[value=-10]() - %1 : int = prim::Constant[value=1]() - %s.2 : int = aten::size(%0, %1) - %3 : None = prim::Constant() - %4 : Device = prim::Constant[value="cuda"]() - %5 : int = prim::Constant[value=6]() - %6 : Tensor = aten::arange(%s.1, %s.2, %5, %3, %4, %3) - return (%6))IR"; - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({1, 8}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({1, 2}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({1, 5}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({1, 5}, {at::kCUDA})); - baidu::mirana::poros::ArangeConverter arangeconverter; - generate_dy_test_helper(graph_IR, &arangeconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenTensorDynamicDtypeConvertsCorrectly) { - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : bool = prim::Constant[value=0]() - %2 : Device = prim::Constant[value="cuda:0"]() - %3 : int = prim::Constant[value=6]() - %4 : int[] = aten::size(%0) - %5 : Tensor = aten::tensor(%4, %3, %2, %1) - return (%5))IR"; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({11, 2, 1}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({10, 2, 1}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({10, 2, 1}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({10, 2, 1}, {at::kCUDA})); - baidu::mirana::poros::TensorConverter tensorconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = true; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &tensorconverter, - input_data, graph_output, poros_output, &prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenLinspaceScalarTensorConvertsCorrectly) { - // aten::linspace(Scalar start, Scalar end, int? steps=None, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None) -> (Tensor) - // aten::linspace目前只能构造dynamic的单测,非dy的单测会被某些pass变为constant - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %2 : int = prim::Constant[value=0]() - %3 : None = prim::Constant() - %start : int = prim::Constant[value=-10]() - %end : int = prim::Constant[value=100]() - %step : int = aten::size(%0, %2) - %device : Device = prim::Constant[value="cuda"]() - %5 : Tensor = aten::linspace(%start, %end, %step, %3, %3, %device, %3) - %6 : Tensor = aten::mul(%0, %5) - return (%6))IR"; - - std::vector input_data; - input_data.push_back(at::ones({6}, {at::kCUDA})); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::ones({10}, {at::kCUDA})); - prewarm_data[1].push_back(at::ones({6}, {at::kCUDA})); - prewarm_data[2].push_back(at::ones({6}, {at::kCUDA})); - - baidu::mirana::poros::LinspaceConverter linspaceconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = true; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &linspaceconverter, - input_data, graph_output, poros_output, &prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenLinspaceStartEndDiffTypeConvertsCorrectly) { - // aten::linspace(Scalar start, Scalar end, int? steps=None, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None) -> (Tensor) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %2 : int = prim::Constant[value=0]() - %3 : None = prim::Constant() - %start : int = prim::Constant[value=-10]() - %end : float = prim::Constant[value=43.3]() - %step : int = aten::size(%0, %2) - %device : Device = prim::Constant[value="cuda"]() - %5 : Tensor = aten::linspace(%start, %end, %step, %3, %3, %device, %3) - %6 : Tensor = aten::mul(%0, %5) - return (%6))IR"; - - std::vector input_data; - input_data.push_back(at::ones({6}, {at::kCUDA})); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::ones({10}, {at::kCUDA})); - prewarm_data[1].push_back(at::ones({6}, {at::kCUDA})); - prewarm_data[2].push_back(at::ones({6}, {at::kCUDA})); - - baidu::mirana::poros::LinspaceConverter linspaceconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = true; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &linspaceconverter, - input_data, graph_output, poros_output, &prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenLinspaceStepNoneConvertsCorrectly) { - std::string graph_IR_str; - if (TORCH_VERSION_MAJOR < 2 && TORCH_VERSION_MINOR < 11) { - // aten::linspace(Scalar start, Scalar end, int? steps=None, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None) -> (Tensor) - graph_IR_str = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : int = prim::Constant[value=0]() - %3 : None = prim::Constant() - %start : int = aten::size(%0, %2) - %end : float = prim::Constant[value=43.3]() - %device : Device = prim::Constant[value="cuda"]() - %5 : Tensor = aten::linspace(%start, %end, %3, %3, %3, %device, %3) - %6 : Tensor = aten::mul(%1, %5) - return (%6))IR"; - } else { - // aten::linspace(Scalar start, Scalar end, int steps, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None) -> (Tensor) - graph_IR_str = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : int = prim::Constant[value=0]() - %3 : None = prim::Constant() - %start : int = aten::size(%0, %2) - %end : float = prim::Constant[value=43.3]() - %step : int = prim::Constant[value=100]() - %device : Device = prim::Constant[value="cuda"]() - %5 : Tensor = aten::linspace(%start, %end, %step, %3, %3, %device, %3) - %6 : Tensor = aten::mul(%1, %5) - return (%6))IR"; - } - const std::string graph_IR = graph_IR_str; - - std::vector input_data; - input_data.push_back(at::ones({1}, {at::kCUDA})); - input_data.push_back(at::ones({100}, {at::kCUDA})); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::ones({6}, {at::kCUDA})); - prewarm_data[0].push_back(at::ones({100}, {at::kCUDA})); - prewarm_data[1].push_back(at::ones({1}, {at::kCUDA})); - prewarm_data[1].push_back(at::ones({100}, {at::kCUDA})); - prewarm_data[2].push_back(at::ones({1}, {at::kCUDA})); - prewarm_data[2].push_back(at::ones({100}, {at::kCUDA})); - - baidu::mirana::poros::LinspaceConverter linspaceconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = true; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &linspaceconverter, - input_data, graph_output, poros_output, &prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenFulllikeConvertsCorrectly) { - // aten::full_like(Tensor self, Scalar fill_value, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None, int? memory_format=None) -> (Tensor) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : None = prim::Constant() - %scalar : float = prim::Constant[value=2.5]() - %out : Tensor = aten::full_like(%0, %scalar, %1, %1, %1, %1, %1) - return (%out))IR"; - std::vector input_data; - input_data.push_back(at::randn({2, 3, 4}, {at::kCUDA})); - baidu::mirana::poros::FulllikeConverter fulllikeconverter; - generate_dy_test_helper(graph_IR, &fulllikeconverter, input_data); -} - -TEST(Converters, ATenFulllikeDefaultTypeConvertsCorrectly) { - // aten::full_like(Tensor self, Scalar fill_value, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None, int? memory_format=None) -> (Tensor) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : None = prim::Constant() - %scalar : float = prim::Constant[value=2.5]() - %out : Tensor = aten::full_like(%0, %scalar, %1, %1, %1, %1, %1) - return (%out))IR"; - std::vector input_data; - auto options_pyt_int = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kInt); - input_data.push_back(at::zeros({2, 3, 4}, options_pyt_int)); - baidu::mirana::poros::FulllikeConverter fulllikeconverter; - generate_dy_test_helper(graph_IR, &fulllikeconverter, input_data); -} - -TEST(Converters, ATenFulllikeDtypeConvertsCorrectly) { - // aten::full_like(Tensor self, Scalar fill_value, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None, int? memory_format=None) -> (Tensor) - // scalar type index in aten and support situation ('o' is support and 'x' is not support): - // uint8_t -> 0 x - // int8_t -> 1 x - // int16_t -> 2 x - // int -> 3 o - // int64_t -> 4 x - // Half -> 5 o - // float -> 6 o - // bool -> 11 x - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : None = prim::Constant() - %2 : int = prim::Constant[value=6]() - %scalar : int = prim::Constant[value=2]() - %out : Tensor = aten::full_like(%0, %scalar, %2, %1, %1, %1, %1) - return (%out))IR"; - std::vector input_data; - input_data.push_back(at::randn({2, 3, 4}, {at::kCUDA})); - baidu::mirana::poros::FulllikeConverter fulllikeconverter; - generate_dy_test_helper(graph_IR, &fulllikeconverter, input_data); -} - -TEST(Converters, ATenFulllikeDynamicConvertsCorrectly) { - // aten::full_like(Tensor self, Scalar fill_value, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None, int? memory_format=None) -> (Tensor) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : None = prim::Constant() - %scalar : int = prim::Constant[value=2]() - %out : Tensor = aten::full_like(%0, %scalar, %1, %1, %1, %1, %1) - return (%out))IR"; - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({4, 5, 6}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({2, 3, 4}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({2, 3, 4}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({2, 3, 4}, {at::kCUDA})); - baidu::mirana::poros::FulllikeConverter fulllikeconverter; - generate_dy_test_helper(graph_IR, &fulllikeconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenFulllikeDynamicDtypeConvertsCorrectly) { - // aten::full_like(Tensor self, Scalar fill_value, *, int? dtype=None, int? layout=None, Device? device=None, bool? pin_memory=None, int? memory_format=None) -> (Tensor) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : None = prim::Constant() - %2 : int = prim::Constant[value=3]() - %scalar : float = prim::Constant[value=2.5]() - %out : Tensor = aten::full_like(%0, %scalar, %2, %1, %1, %1, %1) - return (%out))IR"; - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({4, 5, 6}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({2, 3, 4}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({2, 3, 4}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({2, 3, 4}, {at::kCUDA})); - baidu::mirana::poros::FulllikeConverter fulllikeconverter; - generate_dy_test_helper(graph_IR, &fulllikeconverter, input_data, true, &prewarm_data); -} \ No newline at end of file diff --git a/poros/unittest/converter/group_norm_test.cpp b/poros/unittest/converter/group_norm_test.cpp deleted file mode 100644 index 7dafc83652a..00000000000 --- a/poros/unittest/converter/group_norm_test.cpp +++ /dev/null @@ -1,188 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file group_norm_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/group_norm.h" -#include "poros/util/test_util.h" - -static void groupnorm_test_helper(const std::string& graph_IR, - std::vector& input_data) { - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - baidu::mirana::poros::GroupNormConverter groupnormconverter; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &groupnormconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenGroupNormConvertsCorrectly) { - // aten::group_norm(Tensor input, int num_groups, Tensor? weight=None, Tensor? bias=None, float eps=1e-05, bool cudnn_enabled=True) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor, - %gamma : Tensor, - %beta : Tensor): - %1: int = prim::Constant[value=2]() - %7 : bool = prim::Constant[value=0]() - %8 : float = prim::Constant[value=1.0000000000000001e-05]() - %9 : Tensor = aten::group_norm(%0, %1, %gamma, %beta, %8, %7) - return (%9))IR"; - std::vector input_data; - input_data.push_back(at::randn({2, 10, 3, 3}, {at::kCUDA})); - input_data.push_back(at::randn({10}, {at::kCUDA})); - input_data.push_back(at::randn({10}, {at::kCUDA})); - groupnorm_test_helper(graph_IR, input_data); -} - -TEST(Converters, ATenGroupNormConvertsCorrectly2InputsGamma) { - // aten::group_norm(Tensor input, int num_groups, Tensor? weight=None, Tensor? bias=None, float eps=1e-05, bool cudnn_enabled=True) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor, %gamma : Tensor): - %1 : int = prim::Constant[value=20]() - %2 : None = prim::Constant() - %7 : bool = prim::Constant[value=0]() - %8 : float = prim::Constant[value=1.0000000000000001e-05]() - %9 : Tensor = aten::group_norm(%0, %1, %gamma, %2, %8, %7) - return (%9))IR"; - std::vector input_data; - input_data.push_back(at::randn({4, 100, 50, 50}, {at::kCUDA})); - input_data.push_back(at::randn({100}, {at::kCUDA})); - groupnorm_test_helper(graph_IR, input_data); -} - -TEST(Converters, ATenGroupNormConvertsCorrectlyOneInput) { - // aten::group_norm(Tensor input, int num_groups, Tensor? weight=None, Tensor? bias=None, float eps=1e-05, bool cudnn_enabled=True) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=20]() - %2 : None = prim::Constant() - %7 : bool = prim::Constant[value=0]() - %8 : float = prim::Constant[value=1.0000000000000001e-05]() - %9 : Tensor = aten::group_norm(%0, %1, %2, %2, %8, %7) - return (%9))IR"; - std::vector input_data; - input_data.push_back(at::randn({4, 100, 50, 50}, {at::kCUDA})); - groupnorm_test_helper(graph_IR, input_data); -} - - -static void groupnorm_dy_test_helper(const std::string& graph_IR, - const std::vector& input_data, - bool is_dynamic = false, - std::vector>* prewarm_data = nullptr) { - baidu::mirana::poros::GroupNormConverter groupnormconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = is_dynamic; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &groupnormconverter, - input_data, graph_output, poros_output, prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenGroupNormConvertsDynamicCorrectly) { - // aten::group_norm(Tensor input, int num_groups, Tensor? weight=None, Tensor? bias=None, float eps=1e-05, bool cudnn_enabled=True) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor, - %gamma : Tensor, - %beta : Tensor): - %1: int = prim::Constant[value=2]() - %7 : bool = prim::Constant[value=0]() - %8 : float = prim::Constant[value=1.0000000000000001e-05]() - %9 : Tensor = aten::group_norm(%0, %1, %gamma, %beta, %8, %7) - return (%9))IR"; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({5, 10, 3, 3}, {at::kCUDA})); - prewarm_data[0].push_back(at::ones({10}, {at::kCUDA})); - prewarm_data[0].push_back(at::ones({10}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({2, 10, 3, 3}, {at::kCUDA})); - prewarm_data[1].push_back(at::ones({10}, {at::kCUDA})); - prewarm_data[1].push_back(at::ones({10}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({2, 10, 3, 3}, {at::kCUDA})); - prewarm_data[2].push_back(at::ones({10}, {at::kCUDA})); - prewarm_data[2].push_back(at::ones({10}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({2, 10, 3, 3}, {at::kCUDA})); - input_data.push_back(at::ones({10}, {at::kCUDA})); - input_data.push_back(at::ones({10}, {at::kCUDA})); - - groupnorm_dy_test_helper(graph_IR, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenGroupNormConvertsCorrectlyDynamic2Inputsgamma) { - // aten::group_norm(Tensor input, int num_groups, Tensor? weight=None, Tensor? bias=None, float eps=1e-05, bool cudnn_enabled=True) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor, %gamma : Tensor): - %1 : int = prim::Constant[value=2]() - %2 : None = prim::Constant() - %7 : bool = prim::Constant[value=0]() - %8 : float = prim::Constant[value=1.0000000000000001e-05]() - %9 : Tensor = aten::group_norm(%0, %1, %gamma, %2, %8, %7) - return (%9))IR"; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({20, 100, 50, 50}, {at::kCUDA})); - prewarm_data[0].push_back(at::ones({100}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({10, 100, 40, 40}, {at::kCUDA})); - prewarm_data[1].push_back(at::ones({100}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({10, 100, 40, 40}, {at::kCUDA})); - prewarm_data[2].push_back(at::ones({100}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({10, 100, 40, 40}, {at::kCUDA})); - input_data.push_back(at::ones({100}, {at::kCUDA})); - - groupnorm_dy_test_helper(graph_IR, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenGroupNormConvertsDynamicOneInputCorrectly) { - // aten::group_norm(Tensor input, int num_groups, Tensor? weight=None, Tensor? bias=None, float eps=1e-05, bool cudnn_enabled=True) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=2]() - %2 : None = prim::Constant() - %7 : bool = prim::Constant[value=0]() - %8 : float = prim::Constant[value=1.0000000000000001e-05]() - %9 : Tensor = aten::group_norm(%0, %1, %2, %2, %8, %7) - return (%9))IR"; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({5, 10, 6, 6}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({2, 10, 3, 3}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({2, 10, 3, 3}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({2, 10, 3, 3}, {at::kCUDA})); - - groupnorm_dy_test_helper(graph_IR, input_data, true, &prewarm_data); -} \ No newline at end of file diff --git a/poros/unittest/converter/interpolate_test.cpp b/poros/unittest/converter/interpolate_test.cpp deleted file mode 100644 index 7265c7efcda..00000000000 --- a/poros/unittest/converter/interpolate_test.cpp +++ /dev/null @@ -1,273 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file interpolate_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/interpolate.h" -#include "poros/util/test_util.h" - -static void interpolate_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - std::vector shape){ - std::vector input_data; - input_data.push_back(at::randn(shape, {at::kCUDA})); - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -static std::string gen_upsample_nearest_nd_graph(bool vec_scales, - const std::string& op, - const std::string& output_size, - const std::string& scales) { - std::string output_ir(""); - std::string scales_ir(""); - std::string op_ir(""); - if (!vec_scales) { - output_ir = "int[] = prim::Constant[value=[" + output_size + "]]()"; - if (scales.empty()) { - scales_ir = "None = prim::Constant()"; - } else { - scales_ir = "float = prim::Constant[value=" + scales + "]()"; - } - if (op == "upsample_nearest1d") { - op_ir = op + "(%0, %1, %2)"; - } else if (op == "upsample_nearest2d") { - op_ir = op + "(%0, %1, %2, %2)"; - } else if (op == "upsample_nearest3d") { - op_ir = op + "(%0, %1, %2, %2, %2)"; - } else { - return ""; - } - } else { - if (output_size.empty()) { - output_ir = "None = prim::Constant()"; - } else { - output_ir = "int[] = prim::Constant[value=[" + output_size + "]]()"; - } - if (scales.empty()) { - scales_ir = "None = prim::Constant()"; - } else { - scales_ir = "float[] = prim::Constant[value=[" + scales + "]]()"; - } - op_ir = op + "(%0, %1, %2)"; - } - return R"IR( - graph(%0 : Tensor): - %1 : )IR" + output_ir + R"IR( - %2 : )IR" + scales_ir + R"IR( - %3 : Tensor = aten::)IR" + op_ir + R"IR( - return (%3))IR"; -} - -static std::string gen_upsample_linear_graph(bool vec_scales, - const std::string& op, - const std::string& output_size, - const std::string& align_corners, - const std::string& scales) { - std::string output_ir(""); - std::string scales_ir(""); - std::string op_ir(""); - if (!vec_scales) { - output_ir = "int[] = prim::Constant[value=[" + output_size + "]]()"; - if (scales.empty()) { - scales_ir = "None = prim::Constant()"; - } else { - scales_ir = "float = prim::Constant[value=" + scales + "]()"; - } - if (op == "upsample_linear1d") { - op_ir = op + "(%0, %1, %2, %3)"; - } else if (op == "upsample_bilinear2d") { - op_ir = op + "(%0, %1, %2, %3, %3)"; - } else if (op == "upsample_trilinear3d") { - op_ir = op + "(%0, %1, %2, %3, %3, %3)"; - } else { - return ""; - } - } else { - if (output_size.empty()) { - output_ir = "None = prim::Constant()"; - } else { - output_ir = "int[] = prim::Constant[value=[" + output_size + "]]()"; - } - if (scales.empty()) { - scales_ir = "None = prim::Constant()"; - } else { - scales_ir = "float[] = prim::Constant[value=[" + scales + "]]()"; - } - op_ir = op + "(%0, %1, %2, %3)"; - } - - return R"IR( - graph(%0 : Tensor): - %1 : )IR" + output_ir + R"IR( - %2 : bool = prim::Constant[value=)IR" + align_corners + R"IR(]() - %3 : )IR" + scales_ir + R"IR( - %4 : Tensor = aten::)IR" + op_ir + R"IR( - return (%4))IR"; -} - -TEST(Converters, ATenUpsampleNearest1d) { - // aten::upsample_nearest1d(Tensor self, int[1] output_size, float? scales=None) -> Tensor - const auto graph_IR = gen_upsample_nearest_nd_graph(false, "upsample_nearest1d", "10", ""); - baidu::mirana::poros::UnsampleNearest1DConverter unsamplenearest1dconverter; - interpolate_test_helper(graph_IR, &unsamplenearest1dconverter, {10, 2, 2}); -} - -TEST(Converters, ATenUpsampleNearest1dScalar) { - // aten::upsample_nearest1d(Tensor self, int[1] output_size, float? scales=None) -> Tensor - const auto graph_IR = gen_upsample_nearest_nd_graph(false, "upsample_nearest1d", "8", "4.0"); - baidu::mirana::poros::UnsampleNearest1DConverter unsamplenearest1dconverter; - interpolate_test_helper(graph_IR, &unsamplenearest1dconverter, {10, 2, 2}); -} - -TEST(Converters, ATenUpsampleNearest1dVecScalar) { - // aten::upsample_nearest1d.vec(Tensor input, int[]? output_size, float[]? scale_factors) -> Tensor - const auto graph_IR = gen_upsample_nearest_nd_graph(true, "upsample_nearest1d", "", "4.0"); - baidu::mirana::poros::UnsampleNearest1DConverter unsamplenearest1dconverter; - interpolate_test_helper(graph_IR, &unsamplenearest1dconverter, {10, 2, 2}); -} - -TEST(Converters, ATenUpsampleNearest2d) { - // aten::upsample_nearest2d(Tensor self, int[2] output_size, float? scales_h=None, float? scales_w=None) -> Tensor - const auto graph_IR = gen_upsample_nearest_nd_graph(false, "upsample_nearest2d", "10, 8", ""); - baidu::mirana::poros::UnsampleNearest2DConverter unsamplenearest2dconverter; - interpolate_test_helper(graph_IR, &unsamplenearest2dconverter, {10, 2, 2, 2}); -} - -TEST(Converters, ATenUpsampleNearest2dScalar) { - // aten::upsample_nearest2d(Tensor self, int[2] output_size, float? scales_h=None, float? scales_w=None) -> Tensor - const auto graph_IR = gen_upsample_nearest_nd_graph(false, "upsample_nearest2d", "8, 8", "4.0"); - baidu::mirana::poros::UnsampleNearest2DConverter unsamplenearest2dconverter; - interpolate_test_helper(graph_IR, &unsamplenearest2dconverter, {10, 2, 2, 2}); -} - -TEST(Converters, ATenUpsampleNearest2dVecScalar) { - // aten::upsample_nearest2d.vec(Tensor input, int[]? output_size, float[]? scale_factors) -> Tensor - const auto graph_IR = gen_upsample_nearest_nd_graph(true, "upsample_nearest2d", "", "5.0, 4.0"); - baidu::mirana::poros::UnsampleNearest2DConverter unsamplenearest2dconverter; - interpolate_test_helper(graph_IR, &unsamplenearest2dconverter, {10, 2, 2, 2}); -} - -TEST(Converters, ATenUpsampleNearest3d) { - // aten::upsample_nearest3d(Tensor self, int[3] output_size, float? scales_d=None, float? scales_h=None, float? scales_w=None) -> Tensor - const auto graph_IR = gen_upsample_nearest_nd_graph(false, "upsample_nearest3d", "10, 8, 6", ""); - baidu::mirana::poros::UnsampleNearest3DConverter unsamplenearest3dconverter; - interpolate_test_helper(graph_IR, &unsamplenearest3dconverter, {10, 2, 2, 2, 2}); -} - -TEST(Converters, ATenUpsampleNearest3dScalar) { - // aten::upsample_nearest3d(Tensor self, int[3] output_size, float? scales_d=None, float? scales_h=None, float? scales_w=None) -> Tensor - const auto graph_IR = gen_upsample_nearest_nd_graph(false, "upsample_nearest3d", "8, 8, 8", "4.0"); - baidu::mirana::poros::UnsampleNearest3DConverter unsamplenearest3dconverter; - interpolate_test_helper(graph_IR, &unsamplenearest3dconverter, {10, 2, 2, 2, 2}); -} - -TEST(Converters, ATenUpsampleNearest3dVecScalar) { - // aten::upsample_nearest3d.vec(Tensor input, int[]? output_size, float[]? scale_factors) -> Tensor - const auto graph_IR = gen_upsample_nearest_nd_graph(true, "upsample_nearest3d", "", "5.0, 4.0, 3.0"); - baidu::mirana::poros::UnsampleNearest3DConverter unsamplenearest3dconverter; - interpolate_test_helper(graph_IR, &unsamplenearest3dconverter, {10, 2, 2, 2, 2}); -} - -// start almost equal -TEST(Converters, ATenUpsampleLinear1dWithAlignCorners) { - // aten::upsample_linear1d(Tensor self, int[1] output_size, bool align_corners, float? scales=None) -> Tensor - const auto graph_IR = gen_upsample_linear_graph(false, "upsample_linear1d", "10", "1", ""); - baidu::mirana::poros::UnsampleLinear1DConverter unsamplelinear1dconverter; - interpolate_test_helper(graph_IR, &unsamplelinear1dconverter, {10, 2, 2}); -} - -TEST(Converters, ATenUpsampleLinear1dWithoutAlignCorners) { - // aten::upsample_linear1d(Tensor self, int[1] output_size, bool align_corners, float? scales=None) -> Tensor - const auto graph_IR = gen_upsample_linear_graph(false, "upsample_linear1d", "10", "0", "5.0"); - baidu::mirana::poros::UnsampleLinear1DConverter unsamplelinear1dconverter; - interpolate_test_helper(graph_IR, &unsamplelinear1dconverter, {10, 2, 2}); -} - -TEST(Converters, ATenUpsampleLinear1dScalesWithoutAlignCorners) { - // aten::upsample_linear1d(Tensor self, int[1] output_size, bool align_corners, float? scales=None) -> Tensor - const auto graph_IR = gen_upsample_linear_graph(false, "upsample_linear1d", "8", "0", "4.0"); - baidu::mirana::poros::UnsampleLinear1DConverter unsamplelinear1dconverter; - interpolate_test_helper(graph_IR, &unsamplelinear1dconverter, {10, 2, 2}); -} -TEST(Converters, ATenUpsampleLinear1dVecScaleFactorsWithoutAlignCorners) { - // aten::upsample_linear1d.vec(Tensor input, int[]? output_size, bool align_corners, float[]? scale_factors) -> Tensor - const auto graph_IR = gen_upsample_linear_graph(true, "upsample_linear1d", "", "0", "4.0"); - baidu::mirana::poros::UnsampleLinear1DConverter unsamplelinear1dconverter; - interpolate_test_helper(graph_IR, &unsamplelinear1dconverter, {10, 2, 2}); -} - -TEST(Converters, ATenUpsampleBilinear2dWithAlignCorners) { - // aten::upsample_bilinear2d(Tensor self, int[2] output_size, bool align_corners, float? scales_h=None, float? scales_w=None) -> Tensor - const auto graph_IR = gen_upsample_linear_graph(false, "upsample_bilinear2d", "10, 8", "1", ""); - baidu::mirana::poros::UnsampleBilinear2DConverter unsamplebilinear2dconverter; - interpolate_test_helper(graph_IR, &unsamplebilinear2dconverter, {10, 2, 2, 2}); -} - -TEST(Converters, ATenUpsampleBilinear2dWithoutAlignCorners) { - // aten::upsample_bilinear2d(Tensor self, int[2] output_size, bool align_corners, float? scales_h=None, float? scales_w=None) -> Tensor - const auto graph_IR = gen_upsample_linear_graph(false, "upsample_bilinear2d", "10, 8", "0", ""); - baidu::mirana::poros::UnsampleBilinear2DConverter unsamplebilinear2dconverter; - interpolate_test_helper(graph_IR, &unsamplebilinear2dconverter, {10, 2, 2, 2}); -} - -TEST(Converters, ATenUpsampleBilinear2dScalesWithoutAlignCorners) { - // aten::upsample_bilinear2d(Tensor self, int[2] output_size, bool align_corners, float? scales_h=None, float? scales_w=None) -> Tensor - const auto graph_IR = gen_upsample_linear_graph(false, "upsample_bilinear2d", "10, 10", "0", "5.0"); - baidu::mirana::poros::UnsampleBilinear2DConverter unsamplebilinear2dconverter; - interpolate_test_helper(graph_IR, &unsamplebilinear2dconverter, {10, 2, 2, 2}); -} - -TEST(Converters, ATenUpsampleBilinear2dVecScaleFactorsWithoutAlignCorners) { - // aten::upsample_bilinear2d.vec(Tensor input, int[]? output_size, bool align_corners, float[]? scale_factors) -> Tensor - const auto graph_IR = gen_upsample_linear_graph(true, "upsample_bilinear2d", "", "0", "5.0, 4.0"); - baidu::mirana::poros::UnsampleBilinear2DConverter unsamplebilinear2dconverter; - interpolate_test_helper(graph_IR, &unsamplebilinear2dconverter, {10, 2, 2, 2}); -} - -TEST(Converters, ATenUpsampleTrilinear3dWithAlignCorners) { - // aten::upsample_trilinear3d(Tensor self, int[3] output_size, bool align_corners, float? scales_d=None, float? scales_h=None, float? scales_w=None) -> Tensor - const auto graph_IR = gen_upsample_linear_graph(false, "upsample_trilinear3d", "10, 8, 6", "1", ""); - baidu::mirana::poros::UnsampleTrilinear3DConverter unsampletrilinear3dconverter; - interpolate_test_helper(graph_IR, &unsampletrilinear3dconverter, {10, 2, 2, 2, 2}); -} - -TEST(Converters, ATenUpsampleTrilinear3dWithoutAlignCorners) { - // aten::upsample_trilinear3d(Tensor self, int[3] output_size, bool align_corners, float? scales_d=None, float? scales_h=None, float? scales_w=None) -> Tensor - const auto graph_IR = gen_upsample_linear_graph(false, "upsample_trilinear3d", "10, 8, 6", "0", ""); - baidu::mirana::poros::UnsampleTrilinear3DConverter unsampletrilinear3dconverter; - interpolate_test_helper(graph_IR, &unsampletrilinear3dconverter, {10, 2, 2, 2, 2}); -} - -TEST(Converters, ATenUpsampleTrilinear3dVecScaleFactorsWithoutAlignCorners) { - // aten::upsample_trilinear3d.vec(Tensor input, int[]? output_size, bool align_corners, float[]? scale_factors) -> Tensor - const auto graph_IR = gen_upsample_linear_graph(true, "upsample_trilinear3d", "", "0", "5.0, 4.0, 3.0"); - baidu::mirana::poros::UnsampleTrilinear3DConverter unsampletrilinear3dconverter; - interpolate_test_helper(graph_IR, &unsampletrilinear3dconverter, {10, 2, 2, 2, 2}); -} \ No newline at end of file diff --git a/poros/unittest/converter/layer_norm_test.cpp b/poros/unittest/converter/layer_norm_test.cpp deleted file mode 100644 index fa1ea9f0a3d..00000000000 --- a/poros/unittest/converter/layer_norm_test.cpp +++ /dev/null @@ -1,198 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file layer_norm_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/layer_norm.h" -#include "poros/util/test_util.h" - -static void layernorm_test_helper(const std::string& graph_IR, - std::vector& input_data) { - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - baidu::mirana::poros::LayerNormConverter layernormconverter; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &layernormconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenLayerNormConvertsCorrectlyLast3Dims) { - // aten::layer_norm(Tensor input, int[] normalized_shape, Tensor? weight=None, Tensor? bias=None, float eps=1e-05, bool cudnn_enable=True) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor, - %gamma : Tensor, - %beta : Tensor): - %1: int = prim::Constant[value=3]() - %2: int = prim::Constant[value=100]() - %3: int = prim::Constant[value=100]() - %4 : int[] = prim::ListConstruct(%1, %2, %3) - %7 : bool = prim::Constant[value=0]() - %8 : float = prim::Constant[value=1.0000000000000001e-05]() - %9 : Tensor = aten::layer_norm(%0, %4, %gamma, %beta, %8, %7) - return (%9))IR"; - std::vector input_data; - input_data.push_back(at::randn({4, 3, 100, 100}, {at::kCUDA})); - input_data.push_back(at::randn({3, 100, 100}, {at::kCUDA})); - input_data.push_back(at::randn({3, 100, 100}, {at::kCUDA})); - layernorm_test_helper(graph_IR, input_data); -} - -// 同conv2d -TEST(Converters, ATenLayerNormConvertsCorrectlyLast2Dims) { - const auto graph_IR = R"IR( - graph(%0 : Tensor, - %gamma : Tensor, - %beta : Tensor): - %2: int = prim::Constant[value=100]() - %3: int = prim::Constant[value=100]() - %4 : int[] = prim::ListConstruct(%2, %3) - %7 : bool = prim::Constant[value=0]() - %8 : float = prim::Constant[value=1.0000000000000001e-05]() - %9 : Tensor = aten::layer_norm(%0, %4, %gamma, %beta, %8, %7) - return (%9))IR"; - std::vector input_data; - input_data.push_back(at::randn({4, 3, 100, 100}, {at::kCUDA})); - input_data.push_back(at::randn({100, 100}, {at::kCUDA})); - input_data.push_back(at::randn({100, 100}, {at::kCUDA})); - layernorm_test_helper(graph_IR, input_data); -} - -TEST(Converters, ATenLayerNormConvertsCorrectlyLast1Dims) { - const auto graph_IR = R"IR( - graph(%0 : Tensor, - %gamma : Tensor, - %beta : Tensor): - %3: int = prim::Constant[value=100]() - %4 : int[] = prim::ListConstruct(%3) - %7 : bool = prim::Constant[value=0]() - %8 : float = prim::Constant[value=1.0000000000000001e-05]() - %9 : Tensor = aten::layer_norm(%0, %4, %gamma, %beta, %8, %7) - return (%9))IR"; - std::vector input_data; - input_data.push_back(at::randn({4, 3, 100, 100}, {at::kCUDA})); - input_data.push_back(at::randn({100}, {at::kCUDA})); - input_data.push_back(at::randn({100}, {at::kCUDA})); - layernorm_test_helper(graph_IR, input_data); -} - -TEST(Converters, ATenLayerNormConvertsCorrectly2InputsGamma) { - const auto graph_IR = R"IR( - graph(%0 : Tensor, - %gamma: Tensor): - %beta: None = prim::Constant() - %1: int = prim::Constant[value=100]() - %4 : int[] = prim::ListConstruct(%1) - %7 : bool = prim::Constant[value=0]() - %8 : float = prim::Constant[value=1.0000000000000001e-05]() - %9 : Tensor = aten::layer_norm(%0, %4, %gamma, %beta, %8, %7) - return (%9))IR"; - std::vector input_data; - input_data.push_back(at::randn({4, 3, 100, 100}, {at::kCUDA})); - input_data.push_back(at::randn({100}, {at::kCUDA})); - layernorm_test_helper(graph_IR, input_data); -} - -static void layernorm_dy_test_helper(const std::string& graph_IR, - const std::vector& input_data, - bool is_dynamic = false, - std::vector>* prewarm_data = nullptr) { - baidu::mirana::poros::LayerNormConverter layernormconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = is_dynamic; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &layernormconverter, - input_data, graph_output, poros_output, prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenLayerNormConvertsCorrectly3dDynamicInput1dNormalizedShape) { - const auto graph_IR = R"IR( - graph(%0 : Tensor, - %gamma: Tensor, - %beta: Tensor): - %1: int = prim::Constant[value=4]() - %4 : int[] = prim::ListConstruct(%1) - %7 : bool = prim::Constant[value=0]() - %8 : float = prim::Constant[value=1.0000000000000001e-05]() - %9 : Tensor = aten::layer_norm(%0, %4, %gamma, %beta, %8, %7) - return (%9))IR"; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({10, 3, 4}, {at::kCUDA})); - prewarm_data[0].push_back(at::ones({4}, {at::kCUDA})); - prewarm_data[0].push_back(at::ones({4}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({5, 3, 4}, {at::kCUDA})); - prewarm_data[1].push_back(at::ones({4}, {at::kCUDA})); - prewarm_data[1].push_back(at::ones({4}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({5, 3, 4}, {at::kCUDA})); - prewarm_data[2].push_back(at::ones({4}, {at::kCUDA})); - prewarm_data[2].push_back(at::ones({4}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({5, 3, 4}, {at::kCUDA})); - input_data.push_back(at::ones({4}, {at::kCUDA})); - input_data.push_back(at::ones({4}, {at::kCUDA})); - - layernorm_dy_test_helper(graph_IR, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenLayerNormConvertsCorrectly3dDynamicInput2dNormalizedShape) { - const auto graph_IR = R"IR( - graph(%0 : Tensor, - %gamma : Tensor, - %beta : Tensor): - %2: int = prim::Constant[value=3]() - %3: int = prim::Constant[value=4]() - %4 : int[] = prim::ListConstruct(%2, %3) - %7 : bool = prim::Constant[value=0]() - %8 : float = prim::Constant[value=1.0000000000000001e-05]() - %9 : Tensor = aten::layer_norm(%0, %4, %gamma, %beta, %8, %7) - return (%9))IR"; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({10, 3, 4}, {at::kCUDA})); - prewarm_data[0].push_back(at::ones({3, 4}, {at::kCUDA})); - prewarm_data[0].push_back(at::ones({3, 4}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({5, 3, 4}, {at::kCUDA})); - prewarm_data[1].push_back(at::ones({3, 4}, {at::kCUDA})); - prewarm_data[1].push_back(at::ones({3, 4}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({5, 3, 4}, {at::kCUDA})); - prewarm_data[2].push_back(at::ones({3, 4}, {at::kCUDA})); - prewarm_data[2].push_back(at::ones({3, 4}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({5, 3, 4}, {at::kCUDA})); - input_data.push_back(at::ones({3, 4}, {at::kCUDA})); - input_data.push_back(at::ones({3, 4}, {at::kCUDA})); - - layernorm_dy_test_helper(graph_IR, input_data, true, &prewarm_data); -} \ No newline at end of file diff --git a/poros/unittest/converter/linear_test.cpp b/poros/unittest/converter/linear_test.cpp deleted file mode 100644 index e5fbf236824..00000000000 --- a/poros/unittest/converter/linear_test.cpp +++ /dev/null @@ -1,104 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file linear_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/linear.h" -#include "poros/util/test_util.h" - -static void linear_test_helper(const std::string& graph_IR, - const std::vector& input_data, - const std::vector replace_const_index) { - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - baidu::mirana::poros::LinearConverter linearconverter; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &linearconverter, - input_data, graph_output, poros_output, nullptr, "", replace_const_index)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -static std::string gen_no_bias_graph() { - std::string graph = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : None = prim::Constant() - %3 : Tensor = aten::linear(%0, %1, %2) - return (%3))IR"; - return graph; -} - -TEST(Converters, ATenLinearNoBiasConvertsCorrectly) { - // aten::linear(Tensor input, Tensor weight, Tensor? bias=None) -> Tensor - const auto graph_IR = gen_no_bias_graph(); - baidu::mirana::poros::LinearConverter linearconverter; - std::vector input_data; - input_data.push_back(at::randn({1, 2}, {at::kCUDA})); - input_data.push_back(at::randn({3, 2}, {at::kCUDA})); // 内部转置 - linear_test_helper(graph_IR, input_data, {}); -} - -TEST(Converters, ATenLinearNoBiasNeedPaddingConvertsCorrectly) { - // aten::linear(Tensor input, Tensor weight, Tensor? bias=None) -> Tensor - const auto graph_IR = gen_no_bias_graph(); - baidu::mirana::poros::LinearConverter linearconverter; - std::vector input_data; - input_data.push_back(at::randn({2, 64, 8}, {at::kCUDA})); - input_data.push_back(at::randn({30, 8}, {at::kCUDA})); // 内部转置 - linear_test_helper(graph_IR, input_data, {}); -} - -TEST(Converters, ATenLinearNoBiasNeedPaddingConstWeightConvertsCorrectly) { - // aten::linear(Tensor input, Tensor weight, Tensor? bias=None) -> Tensor - const auto graph_IR = gen_no_bias_graph(); - baidu::mirana::poros::LinearConverter linearconverter; - std::vector input_data; - input_data.push_back(at::randn({2, 64, 8}, {at::kCUDA})); - input_data.push_back(at::randn({30, 8}, {at::kCUDA})); // 内部转置 - linear_test_helper(graph_IR, input_data, {1}); //把第二个参数转换成常量 -} - -TEST(Converters, ATenLinearNoBiasNeedPaddingConstWeight2ConvertsCorrectly) { - // aten::linear(Tensor input, Tensor weight, Tensor? bias=None) -> Tensor - const auto graph_IR = gen_no_bias_graph(); - baidu::mirana::poros::LinearConverter linearconverter; - std::vector input_data; - input_data.push_back(at::randn({2, 64, 64, 8}, {at::kCUDA})); - input_data.push_back(at::randn({30, 8}, {at::kCUDA})); // 内部转置 - linear_test_helper(graph_IR, input_data, {1}); //把第二个参数转换成常量 -} - -TEST(Converters, ATenLinearBiasConvertsCorrectly) { - // aten::linear(Tensor input, Tensor weight, Tensor? bias=None) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor, %2 : Tensor): - %3 : Tensor = aten::linear(%0, %1, %2) - return (%3))IR"; - baidu::mirana::poros::LinearConverter linearconverter; - std::vector input_data; - input_data.push_back(at::randn({1, 3}, {at::kCUDA})); - input_data.push_back(at::randn({2, 3}, {at::kCUDA})); - input_data.push_back(at::randn({2}, {at::kCUDA})); - linear_test_helper(graph_IR, input_data, {}); -} \ No newline at end of file diff --git a/poros/unittest/converter/logical_test.cpp b/poros/unittest/converter/logical_test.cpp deleted file mode 100644 index 945c4e74181..00000000000 --- a/poros/unittest/converter/logical_test.cpp +++ /dev/null @@ -1,201 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file logical_test.cpp -* @author Lin Xiao Chun (linxiaochun@baidu.com) -* @date 2022-02-17 18:32:15 -* @brief -**/ - -#include -#include - -#include "poros/converter/gpu/logical.h" -#include "poros/util/test_util.h" - -enum InputTypeEnum { - TYPE_A = 0, // [4]*[4] - TYPE_B, // [2,2]*[2,2] - TYPE_C, // [4]*[true] - TYPE_D, //broadcasting [1,3,2]*[2] - TYPE_E, //broadcasting [2,3,4]*[3,4] -}; - -static std::vector get_input_data(const InputTypeEnum input_type) { - std::vector input_data; - auto options_pyt = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kBool); - - switch (input_type) { - case TYPE_A: // [4]*[4] - input_data.push_back(torch::tensor({false, true, false, true}, options_pyt)); - input_data.push_back(torch::tensor({false, true, true, true}, options_pyt)); - break; - case TYPE_B:// [2,2]*[2,2] - input_data.push_back(torch::tensor({{false, true}, - {false, true}}, options_pyt)); - input_data.push_back(torch::tensor({{false, true}, - {true, true}}, options_pyt)); - break; - case TYPE_C:// [4]*[1] - input_data.push_back(torch::tensor({false, true, false, true}, options_pyt)); - input_data.push_back(torch::tensor({true}, options_pyt)); - break; - case TYPE_D://broadcasting [1,3,2]*[2] - input_data.push_back(torch::tensor({{{true, true}, {false, true}, {false, false}}}, options_pyt)); - input_data.push_back(torch::tensor({false, true}, options_pyt)); - break; - case TYPE_E://broadcasting [2,3,4]*[3,4] - input_data.push_back(torch::tensor({ - {{false, true, false, true}, {false, true, false, false}, - {true, true, true, true}}, - {{false, true, false, false}, {true, true, true, true}, - {false, true, false, true}} - }, options_pyt)); - input_data.push_back(torch::tensor({{false, true, false, true}, - {false, true, false, false}, - {true, true, true, true}}, options_pyt)); - break; - } - - return input_data; -} - -static void and_test_helper(const std::string &graph_IR, - baidu::mirana::poros::IConverter *converter, - const InputTypeEnum input_type) { - - auto input_data = get_input_data(input_type); - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -static std::string gen_and_or_tensor_graph(const std::string &op) { - return R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : Tensor = aten::)IR" + op + R"IR((%0, %1) - return (%2))IR"; -} - -static std::string gen_not_tensor_graph(const std::string &op) { - return R"IR( - graph(%0 : Tensor): - %2 : Tensor = aten::)IR" + op + R"IR((%0) - return (%2))IR"; -} - -TEST(Converters, ATenLogicalAndConvertsCorrectly) { - - const auto graph_IR = gen_and_or_tensor_graph("__and__"); - baidu::mirana::poros::AndConverter converter; - and_test_helper(graph_IR, &converter, TYPE_A); - and_test_helper(graph_IR, &converter, TYPE_B); - and_test_helper(graph_IR, &converter, TYPE_C); - and_test_helper(graph_IR, &converter, TYPE_D); - and_test_helper(graph_IR, &converter, TYPE_E); -} - -TEST(Converters, ATenLogicalBitwiseAndConvertsCorrectly) { - - const auto graph_IR = gen_and_or_tensor_graph("bitwise_and"); - baidu::mirana::poros::AndConverter converter; - and_test_helper(graph_IR, &converter, TYPE_A); - and_test_helper(graph_IR, &converter, TYPE_B); - and_test_helper(graph_IR, &converter, TYPE_C); - and_test_helper(graph_IR, &converter, TYPE_D); - and_test_helper(graph_IR, &converter, TYPE_E); -} - -TEST(Converters, ATenLogicalOrConvertsCorrectly) { - - const auto graph_IR = gen_and_or_tensor_graph("__or__"); - baidu::mirana::poros::OrConverter converter; - and_test_helper(graph_IR, &converter, TYPE_A); - and_test_helper(graph_IR, &converter, TYPE_B); - and_test_helper(graph_IR, &converter, TYPE_C); - and_test_helper(graph_IR, &converter, TYPE_D); - and_test_helper(graph_IR, &converter, TYPE_E); -} - -TEST(Converters, ATenLogicalBitwiseOrConvertsCorrectly) { - - const auto graph_IR = gen_and_or_tensor_graph("bitwise_or"); - baidu::mirana::poros::OrConverter converter; - and_test_helper(graph_IR, &converter, TYPE_A); - and_test_helper(graph_IR, &converter, TYPE_B); - and_test_helper(graph_IR, &converter, TYPE_C); - and_test_helper(graph_IR, &converter, TYPE_D); - and_test_helper(graph_IR, &converter, TYPE_E); -} - -TEST(Converters, ATenLogicalXOrConvertsCorrectly) { - - const auto graph_IR = gen_and_or_tensor_graph("__xor__"); - baidu::mirana::poros::XorConverter converter; - and_test_helper(graph_IR, &converter, TYPE_A); - and_test_helper(graph_IR, &converter, TYPE_B); - and_test_helper(graph_IR, &converter, TYPE_C); - and_test_helper(graph_IR, &converter, TYPE_D); - and_test_helper(graph_IR, &converter, TYPE_E); -} - -TEST(Converters, ATenLogicalBitwiseXOrConvertsCorrectly) { - - const auto graph_IR = gen_and_or_tensor_graph("bitwise_xor"); - baidu::mirana::poros::XorConverter converter; - and_test_helper(graph_IR, &converter, TYPE_A); - and_test_helper(graph_IR, &converter, TYPE_B); - and_test_helper(graph_IR, &converter, TYPE_C); - and_test_helper(graph_IR, &converter, TYPE_D); - and_test_helper(graph_IR, &converter, TYPE_E); -} - -static void not_test_helper(const std::string &graph_IR, - baidu::mirana::poros::IConverter *converter, - const InputTypeEnum input_type) { - - auto input_data = get_input_data(input_type); - input_data.pop_back(); // only need one input, pop out the last one - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenLogicalBitwiseNotConvertsCorrectly) { - - const auto graph_IR = gen_not_tensor_graph("bitwise_not"); - baidu::mirana::poros::NotConverter converter; - not_test_helper(graph_IR, &converter, TYPE_A); - not_test_helper(graph_IR, &converter, TYPE_B); - not_test_helper(graph_IR, &converter, TYPE_C); - not_test_helper(graph_IR, &converter, TYPE_D); - not_test_helper(graph_IR, &converter, TYPE_E); -} - -//} \ No newline at end of file diff --git a/poros/unittest/converter/lstm_cell_test.cpp b/poros/unittest/converter/lstm_cell_test.cpp deleted file mode 100644 index 3f04dea455b..00000000000 --- a/poros/unittest/converter/lstm_cell_test.cpp +++ /dev/null @@ -1,75 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file lstm_cell_test.cpp -* @author wangrui39@baidu.com -* @date Mon December 13 11:36:11 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/lstm_cell.h" -#include "poros/util/test_util.h" - -static void linear_test_helper(const std::string& graph_IR, - const std::vector& input_data) { - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - baidu::mirana::poros::LstmCellConverter lstm_cellconverter; - - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &lstm_cellconverter, - input_data, graph_output, poros_output)); - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenlstm_cellconverterCorrectly) { - //aten::lstm_cell(Tensor input, Tensor[] hx, Tensor w_ih, Tensor w_hh, Tensor? b_ih=None, Tensor? b_hh=None) -> (Tensor, Tensor) - - const auto graph = R"IR( - graph(%0 : Tensor, - %1 : Tensor, - %3 : Tensor, - %4 : Tensor, - %5 : Tensor, - %6 : Tensor, - %7 : Tensor): - %2 : Tensor[] = prim::ListConstruct(%0, %1) - %8 : Tensor, %9 : Tensor = aten::lstm_cell(%3, %2, %4, %5, %6, %7) - return (%8))IR"; - - std::vector input_data; - auto input = at::randn({50, 10}, {at::kCUDA}); - auto h0 = at::randn({50, 20}, {at::kCUDA}); - auto c0 = at::randn({50, 20}, {at::kCUDA}); - auto w_ih = at::randn({4 * 20, 10}, {at::kCUDA}); - auto w_hh = at::randn({4 * 20, 20}, {at::kCUDA}); - auto b_ih = at::randn({4 * 20}, {at::kCUDA}); - auto b_hh = at::randn({4 * 20}, {at::kCUDA}); - - input_data.push_back(h0); - input_data.push_back(c0); - input_data.push_back(input); - input_data.push_back(w_ih); - input_data.push_back(w_hh); - input_data.push_back(b_ih); - input_data.push_back(b_hh); - - linear_test_helper(graph, input_data); -} diff --git a/poros/unittest/converter/lstm_test.cpp b/poros/unittest/converter/lstm_test.cpp deleted file mode 100644 index 012e6701a5a..00000000000 --- a/poros/unittest/converter/lstm_test.cpp +++ /dev/null @@ -1,224 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file lstm_cell_test.cpp -* @author wangrui39@baidu.com -* @date Mon December 13 11:36:11 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/lstm.h" -#include "poros/util/test_util.h" - -static void lstm_test_helper(const std::string& graph_IR, - const std::vector& input_data, - baidu::mirana::poros::IConverter* converter) { - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output)); - ASSERT_EQ(3, graph_output.size()); - ASSERT_EQ(3, poros_output.size()); - - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[1], poros_output[1], 2e-6)); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[2], poros_output[2], 2e-6)); - -} - -TEST(Converters, ATenlstmconverterCorrectly) { - // aten::lstm.input(Tensor input, Tensor[] hx, Tensor[] params, bool has_biases, int num_layers, float dropout, bool train, bool bidirectional, bool batch_first) -> (Tensor, Tensor, Tensor) - // num_layers = 1 - // bidirectional = false - // batch_first = false - const auto graph = R"IR( - graph( %0 : Tensor, - %1 : Tensor, - %2 : Tensor, - %3 : Tensor, - %4 : Tensor, - %5 : Tensor, - %6 : Tensor): - %11 : bool = prim::Constant[value=1]() - %12 : bool = prim::Constant[value=0]() - %13 : int = prim::Constant[value=1]() - %14 : float = prim::Constant[value=0.0]() - %15 : Tensor[] = prim::ListConstruct(%0, %1) - %16 : Tensor[] = prim::ListConstruct(%3, %4, %5, %6) - %17 : Tensor, %18 : Tensor, %19 : Tensor = aten::lstm(%2, %15, %16, %11, %13, %14, %12, %12, %12) - return (%17, %18, %19))IR"; - - /*const auto graph = R"IR( - graph( %0 : Tensor, - %1 : Tensor, - %2 : Tensor, - %3 : Tensor, - %4 : Tensor, - %5 : Tensor, - %6 : Tensor): - %11 : bool = prim::Constant[value=1]() - %12 : bool = prim::Constant[value=0]() - %13 : int = prim::Constant[value=1]() - %14 : float = prim::Constant[value=0.0]() - %15 : Tensor[] = prim::ListConstruct(%0, %1) - %16 : Tensor[] = prim::ListConstruct(%3, %4, %5, %6) - %17 : Tensor, %18 : Tensor, %19 : Tensor = aten::lstm(%2, %15, %16, %11, %13, %14, %12, %12, %12) - return (%17, %18, %19))IR";*/ - - std::vector input_data; - auto input = at::randn({1, 5, 1}, {at::kCUDA}); - auto h0 = at::randn({1, 5, 2}, {at::kCUDA}); - auto c0 = at::randn({1, 5, 2}, {at::kCUDA}); - - auto w1 = at::randn({8, 1}, {at::kCUDA}); - auto w2 = at::randn({8, 2}, {at::kCUDA}); - auto w3 = at::randn({8}, {at::kCUDA}); - auto w4 = at::randn({8}, {at::kCUDA}); - - input_data.push_back(h0); - input_data.push_back(c0); - input_data.push_back(input); - - input_data.push_back(w1); - input_data.push_back(w2); - input_data.push_back(w3); - input_data.push_back(w4); - - - baidu::mirana::poros::LstmConverter lstmconverter; - lstm_test_helper(graph, input_data, &lstmconverter); -} - -TEST(Converters, ATenlstmconverterBidirectionalCorrectly) { - // aten::lstm.input(Tensor input, Tensor[] hx, Tensor[] params, bool has_biases, int num_layers, float dropout, bool train, bool bidirectional, bool batch_first) -> (Tensor, Tensor, Tensor) - // num_layers = 1 - // bidirectional = true - // batch_first = true - const auto graph = R"IR( - graph( %0 : Tensor, - %1 : Tensor, - %2 : Tensor, - %3 : Tensor, - %4 : Tensor, - %5 : Tensor, - %6 : Tensor, - %7 : Tensor, - %8 : Tensor, - %9 : Tensor, - %10 : Tensor): - %11 : bool = prim::Constant[value=1]() - %12 : bool = prim::Constant[value=0]() - %13 : int = prim::Constant[value=1]() - %14 : float = prim::Constant[value=0.0]() - %15 : Tensor[] = prim::ListConstruct(%0, %1) - %16 : Tensor[] = prim::ListConstruct(%3, %4, %5, %6, %7, %8, %9, %10) - %17 : Tensor, %18 : Tensor, %19 : Tensor = aten::lstm(%2, %15, %16, %11, %13, %14, %12, %11, %11) - return (%17, %18, %19))IR"; - - - std::vector input_data; - auto input = at::randn({50, 7, 10}, {at::kCUDA}); - auto h0 = at::randn({2, 50, 20}, {at::kCUDA}); - auto c0 = at::randn({2, 50, 20}, {at::kCUDA}); - - auto w1 = at::randn({80, 10}, {at::kCUDA}); - auto w2 = at::randn({80, 20}, {at::kCUDA}); - auto w3 = at::randn({80}, {at::kCUDA}); - auto w4 = at::randn({80}, {at::kCUDA}); - - auto r_w1 = at::randn({80, 10}, {at::kCUDA}); - auto r_w2 = at::randn({80, 20}, {at::kCUDA}); - auto r_w3 = at::randn({80}, {at::kCUDA}); - auto r_w4 = at::randn({80}, {at::kCUDA}); - - input_data.push_back(h0); - input_data.push_back(c0); - input_data.push_back(input); - - input_data.push_back(w1); - input_data.push_back(w2); - input_data.push_back(w3); - input_data.push_back(w4); - input_data.push_back(r_w1); - input_data.push_back(r_w2); - input_data.push_back(r_w3); - input_data.push_back(r_w4); - - - baidu::mirana::poros::LstmConverter lstmconverter; - lstm_test_helper(graph, input_data, &lstmconverter); -} - -TEST(Converters, ATenlstmconverterNumlayerCorrectly) { - // aten::lstm.input(Tensor input, Tensor[] hx, Tensor[] params, bool has_biases, int num_layers, float dropout, bool train, bool bidirectional, bool batch_first) -> (Tensor, Tensor, Tensor) - // num_layers > 1 - // bidirectional = false - // batch_first = true - const auto graph = R"IR( - graph( %0 : Tensor, - %1 : Tensor, - %2 : Tensor, - %3 : Tensor, - %4 : Tensor, - %5 : Tensor, - %6 : Tensor, - %7 : Tensor, - %8 : Tensor, - %9 : Tensor, - %10 : Tensor): - %11 : bool = prim::Constant[value=1]() - %12 : bool = prim::Constant[value=0]() - %13 : int = prim::Constant[value=2]() - %14 : float = prim::Constant[value=0.0]() - %15 : Tensor[] = prim::ListConstruct(%0, %1) - %16 : Tensor[] = prim::ListConstruct(%3, %4, %5, %6, %7, %8, %9, %10) - %17 : Tensor, %18 : Tensor, %19 : Tensor = aten::lstm(%2, %15, %16, %11, %13, %14, %12, %12, %11) - return (%17, %18, %19))IR"; - - std::vector input_data; - auto input = at::randn({50, 7, 10}, {at::kCUDA}); - auto h0 = at::randn({2, 50, 20}, {at::kCUDA}); - auto c0 = at::randn({2, 50, 20}, {at::kCUDA}); - - auto num1_w1 = at::randn({80, 10}, {at::kCUDA}); - auto num1_w2 = at::randn({80, 20}, {at::kCUDA}); - auto num1_w3 = at::randn({80}, {at::kCUDA}); - auto num1_w4 = at::randn({80}, {at::kCUDA}); - auto num2_w1 = at::randn({80, 20}, {at::kCUDA}); - auto num2_w2 = at::randn({80, 20}, {at::kCUDA}); - auto num2_w3 = at::randn({80}, {at::kCUDA}); - auto num2_w4 = at::randn({80}, {at::kCUDA}); - - input_data.push_back(h0); - input_data.push_back(c0); - input_data.push_back(input); - - input_data.push_back(num1_w1); - input_data.push_back(num1_w2); - input_data.push_back(num1_w3); - input_data.push_back(num1_w4); - input_data.push_back(num2_w1); - input_data.push_back(num2_w2); - input_data.push_back(num2_w3); - input_data.push_back(num2_w4); - - baidu::mirana::poros::LstmConverter lstmconverter; - lstm_test_helper(graph, input_data, &lstmconverter); -} diff --git a/poros/unittest/converter/matrix_multiply_test.cpp b/poros/unittest/converter/matrix_multiply_test.cpp deleted file mode 100644 index f96307148d3..00000000000 --- a/poros/unittest/converter/matrix_multiply_test.cpp +++ /dev/null @@ -1,104 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file matrix_multiply_test.cpp -* @author tianjinjin@baidu.com -* @date Tue Sep 14 18:19:00 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/matrix_multiply.h" -#include "poros/util/test_util.h" - -static void matrix_multiply_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - std::vector shape1, - std::vector shape2, - bool tripleinputs = false, - std::vector shape3 = {5}) { - std::vector input_data; - input_data.push_back(at::randn(shape1, {at::kCUDA})); - input_data.push_back(at::randn(shape2, {at::kCUDA})); - if (tripleinputs){ - input_data.push_back(at::randn(shape3, {at::kCUDA})); - } - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - c10::ShowLogInfoToStderr(); - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenMatmulConvertersCorrectly) { - // aten::matmul(Tensor self, Tensor other) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : Tensor = aten::matmul(%0, %1) - return (%2))IR"; - baidu::mirana::poros::MatmulConverter matmulconverter; - matrix_multiply_test_helper(graph_IR, &matmulconverter, {3}, {3}); - matrix_multiply_test_helper(graph_IR, &matmulconverter, {1, 1536}, {1536, 2}); - matrix_multiply_test_helper(graph_IR, &matmulconverter, {3}, {3, 512}); - matrix_multiply_test_helper(graph_IR, &matmulconverter, {512}, {512, 3}); - matrix_multiply_test_helper(graph_IR, &matmulconverter, {512, 3}, {3}); - matrix_multiply_test_helper(graph_IR, &matmulconverter, {1, 30, 1024}, {1024}); - matrix_multiply_test_helper(graph_IR, &matmulconverter, {1, 30, 1024}, {1024, 214}); - matrix_multiply_test_helper(graph_IR, &matmulconverter, {8}, {512, 8, 10}); - matrix_multiply_test_helper(graph_IR, &matmulconverter, {254, 8}, {512, 8, 10}); - matrix_multiply_test_helper(graph_IR, &matmulconverter, {10, 3, 512}, {10, 512, 214}); - matrix_multiply_test_helper(graph_IR, &matmulconverter, {10, 1, 24, 224}, {7, 224, 5}); -} - -TEST(Converters, ATenBmmConvertersCorrectly) { - // aten::bmm(Tensor self, Tensor mat2) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : Tensor = aten::bmm(%0, %1) - return (%2))IR"; - baidu::mirana::poros::BmmConverter bmmconverter; - matrix_multiply_test_helper(graph_IR, &bmmconverter, {10, 3, 4}, {10, 4, 5}); -} - -static std::string gen_addmm_graph(const std::string& beta, const std::string& alpha) { - return R"IR( - graph(%0 : Tensor, %1 : Tensor, %2 : Tensor): - %3 : float = prim::Constant[value=)IR" + beta + R"IR(]() - %4 : float = prim::Constant[value=)IR" + alpha + R"IR(]() - %5 : Tensor = aten::addmm(%0, %1, %2, %3, %4) - return (%5))IR"; -} - -TEST(Converters, ATenAddmmConvertersCorrectly) { - // aten::addmm(Tensor self, Tensor mat1, Tensor mat2, *, Scalar beta=1, Scalar alpha=1) -> Tensor - const auto graph_IR = gen_addmm_graph("1.0", "1.0"); - baidu::mirana::poros::AddmmConverter addmmconverter; - matrix_multiply_test_helper(graph_IR, &addmmconverter, {2, 3}, {2, 3}, true, {3, 3}); - matrix_multiply_test_helper(graph_IR, &addmmconverter, {3}, {2, 3}, true, {3, 3}); -} - -TEST(Converters, ATenAddmmBetaAlphaConvertersCorrectly) { - // aten::addmm(Tensor self, Tensor mat1, Tensor mat2, *, Scalar beta=1, Scalar alpha=1) -> Tensor - const auto graph_IR = gen_addmm_graph("3.3", "2.2"); - baidu::mirana::poros::AddmmConverter addmmconverter; - matrix_multiply_test_helper(graph_IR, &addmmconverter, {2, 3}, {2, 3}, true, {3, 3}); -} \ No newline at end of file diff --git a/poros/unittest/converter/meshgrid_test.cpp b/poros/unittest/converter/meshgrid_test.cpp deleted file mode 100644 index 7e2339bc09d..00000000000 --- a/poros/unittest/converter/meshgrid_test.cpp +++ /dev/null @@ -1,66 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file meshgrid_test.cpp -* @author wangrui39@baidu.com -* @date Monday November 27 11:36:11 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/meshgrid.h" -#include "poros/util/test_util.h" - -static void add_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - std::vector value1 = {1.0, 2.0, 3.0}, - std::vector value2 = {4.0, 5.0}){ - std::vector input_data; - auto options_pyt = torch::TensorOptions().device(torch::kCUDA, 0);//.dtype(torch::kInt32); - input_data.push_back(at::tensor(value1, options_pyt)); - input_data.push_back(at::tensor(value2, options_pyt)); - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -static std::string gen_meshgrid_graph() { - std::string graph = R"IR( - graph(%x.1 : Tensor, - %y.1 : Tensor): - %10 : int = prim::Constant[value=1]() - %4 : Tensor[] = prim::ListConstruct(%x.1, %y.1) - %5 : Tensor[] = aten::meshgrid(%4) - %grid_x.1 : Tensor, %grid_y.1 : Tensor = prim::ListUnpack(%5) - %11 : Tensor = aten::add(%grid_x.1, %grid_y.1, %10) - return (%11))IR"; - - return graph; -} - -TEST(Converters, ATenMeshgridConvertsCorrectly) { - const auto graph_IR = gen_meshgrid_graph(); - baidu::mirana::poros::MeshgridConverter meshgridconverter; - add_test_helper(graph_IR, &meshgridconverter); -} \ No newline at end of file diff --git a/poros/unittest/converter/mul_div_test.cpp b/poros/unittest/converter/mul_div_test.cpp deleted file mode 100644 index 3e7d8fb1c0d..00000000000 --- a/poros/unittest/converter/mul_div_test.cpp +++ /dev/null @@ -1,410 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file mul_div_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/mul_div.h" -#include "poros/util/test_util.h" - -static void mul_div_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - bool singleInput, - std::vector shape1 = {5}, - std::vector shape2 = {5}) { - std::vector input_data; - input_data.push_back(at::randn(shape1, {at::kCUDA})); - if (!singleInput){ - input_data.push_back(at::randn(shape2, {at::kCUDA})); - } - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -std::string gen_mul_div_tensor_graph(const std::string& op) { - return R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : Tensor = aten::)IR" + op + R"IR((%0, %1) - return (%2))IR"; -} - -std::string gen_mul_div_scalar_graph(const std::string& op, const std::string& scalar) { - return R"IR( - graph(%0 : Tensor): - %1 : float = prim::Constant[value=)IR" + scalar + R"IR(]() - %2 : Tensor = aten::)IR" + op + R"IR((%0, %1) - return (%2))IR"; -} - -TEST(Converters, ATenMulConvertsCorrectly) { - // aten::mul.Tensor(Tensor self, Tensor other) -> Tensor - const auto graph_IR = gen_mul_div_tensor_graph("mul"); - baidu::mirana::poros::MulConverter mulconverter; - mul_div_test_helper(graph_IR, &mulconverter, false); - mul_div_test_helper(graph_IR, &mulconverter, false, {3, 4}, {4}); - mul_div_test_helper(graph_IR, &mulconverter, false, {4}, {3, 4}); - mul_div_test_helper(graph_IR, &mulconverter, false, {4, 1}, {1, 4}); - mul_div_test_helper(graph_IR, &mulconverter, false, {3, 4, 3}, {4, 3}); - mul_div_test_helper(graph_IR, &mulconverter, false, {4, 3}, {3, 4, 3}); -} - -TEST(Converters, ATenMulScalarConvertsCorrectly) { - // aten::mul.Scalar(Tensor self, Scalar other) -> Tensor - const auto graph_IR = gen_mul_div_scalar_graph("mul", "2.4"); - baidu::mirana::poros::MulConverter mulconverter; - mul_div_test_helper(graph_IR, &mulconverter, true); - mul_div_test_helper(graph_IR, &mulconverter, true, {3, 4, 3}); -} - -TEST(Converters, ATenMul_ConvertsCorrectly) { - // aten::mul_.Tensor(Tensor(a!) self, Tensor other) -> Tensor(a!) - const auto graph_IR = gen_mul_div_tensor_graph("mul_"); - baidu::mirana::poros::MulConverter mulconverter; - mul_div_test_helper(graph_IR, &mulconverter, false); - mul_div_test_helper(graph_IR, &mulconverter, false, {3, 4}, {4}); - mul_div_test_helper(graph_IR, &mulconverter, false, {3, 4, 3}, {4, 3}); -} - -TEST(Converters, ATenMul_ScalarConvertsCorrectly) { - // aten::mul_.Scalar(Tensor(a!) self, Scalar other) -> Tensor(a!) - const auto graph_IR = gen_mul_div_scalar_graph("mul_", "2.4"); - baidu::mirana::poros::MulConverter mulconverter; - mul_div_test_helper(graph_IR, &mulconverter, true); - mul_div_test_helper(graph_IR, &mulconverter, true, {3, 4, 3}); -} - -TEST(Converters, ATenDivConvertsCorrectly) { - // aten::div.Tensor(Tensor self, Tensor other) -> Tensor - const auto graph_IR = gen_mul_div_tensor_graph("div"); - baidu::mirana::poros::DivConverter divconverter; - mul_div_test_helper(graph_IR, &divconverter, false); - mul_div_test_helper(graph_IR, &divconverter, false, {3, 4}, {4}); - mul_div_test_helper(graph_IR, &divconverter, false, {4}, {3, 4}); - mul_div_test_helper(graph_IR, &divconverter, false, {4, 1}, {1, 4}); - mul_div_test_helper(graph_IR, &divconverter, false, {3, 4, 3}, {4, 3}); - mul_div_test_helper(graph_IR, &divconverter, false, {4, 3}, {3, 4, 3}); -} - -TEST(Converters, ATenDivScalarConvertsCorrectly) { - // aten::div.Scalar(Tensor self, Scalar other) -> (Tensor) - const auto graph_IR = gen_mul_div_scalar_graph("div", "2.4"); - baidu::mirana::poros::DivConverter divconverter; - mul_div_test_helper(graph_IR, &divconverter, true); - mul_div_test_helper(graph_IR, &divconverter, true, {3, 4, 3}); -} - -TEST(Converters, ATenDiv_ConvertsCorrectly) { - // aten::div_.Tensor(Tensor(a!) self, Tensor other) -> Tensor(a!) - const auto graph_IR = gen_mul_div_tensor_graph("div_"); - baidu::mirana::poros::DivConverter divconverter; - mul_div_test_helper(graph_IR, &divconverter, false); - mul_div_test_helper(graph_IR, &divconverter, false, {3, 4}, {4}); - mul_div_test_helper(graph_IR, &divconverter, false, {3, 4, 3}, {4, 3}); -} - -TEST(Converters, ATenDiv_ScalarConvertsCorrectly) { - // aten::div_.Scalar(Tensor(a!) self, Scalar other) -> Tensor(a!) - const auto graph_IR = gen_mul_div_scalar_graph("div_", "2.4"); - baidu::mirana::poros::DivConverter divconverter; - mul_div_test_helper(graph_IR, &divconverter, true); - mul_div_test_helper(graph_IR, &divconverter, true, {3, 4, 3}); -} - -TEST(Converters, ATenDivIntDivideIntConvertsCorrectly) { - // aten::div.Tensor(Tensor self, Tensor other) -> Tensor - const auto graph_IR = gen_mul_div_tensor_graph("div"); - - auto options_pyt_int = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kInt); - std::vector input_data; - input_data.push_back(torch::tensor({14}, options_pyt_int)); - input_data.push_back(torch::tensor({2}, options_pyt_int)); - - baidu::mirana::poros::DivConverter divconverter; - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &divconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenDivFloatDivideIntConvertsCorrectly) { - // aten::div.Scalar(Tensor self, Scalar other) -> (Tensor) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=3]() - %2 : Tensor = aten::div(%0, %1) - return (%2))IR"; - - auto options_pyt_float = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kFloat); - std::vector input_data; - input_data.push_back(torch::tensor({15.3}, options_pyt_float)); - - baidu::mirana::poros::DivConverter divconverter; - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &divconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenDivIntDivideFloatConvertsCorrectly) { - // aten::div.Scalar(Tensor self, Scalar other) -> (Tensor) - const auto graph_IR = gen_mul_div_scalar_graph("div", "2.4"); - - auto options_pyt_int = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kInt); - std::vector input_data; - input_data.push_back(torch::tensor({15}, options_pyt_int)); - - baidu::mirana::poros::DivConverter divconverter; - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &divconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenRemainderConvertsCorrectly) { - // aten::remainder.Tensor(Tensor self, Tensor other) -> Tensor - const auto graph_IR = gen_mul_div_tensor_graph("remainder"); - baidu::mirana::poros::RemainderConverter remainder; - mul_div_test_helper(graph_IR, &remainder, false); - mul_div_test_helper(graph_IR, &remainder, false, {3, 4}, {4}); - mul_div_test_helper(graph_IR, &remainder, false, {4}, {3, 4}); - mul_div_test_helper(graph_IR, &remainder, false, {4, 1}, {1, 4}); - mul_div_test_helper(graph_IR, &remainder, false, {3, 4, 3}, {4, 3}); - mul_div_test_helper(graph_IR, &remainder, false, {4, 3}, {3, 4, 3}); -} - -TEST(Converters, ATenRemainderScalarConvertsCorrectly) { - // aten::remainder.Scalar(Tensor self, Scalar other) -> Tensor - const auto graph_IR = gen_mul_div_scalar_graph("remainder", "-0.4"); - baidu::mirana::poros::RemainderConverter remainder; - mul_div_test_helper(graph_IR, &remainder, true); - mul_div_test_helper(graph_IR, &remainder, true, {3, 4, 3}); -} - - -static void mul_div_dynamic_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - const std::vector& input_data, - bool is_dynamic = false, - std::vector>* prewarm_data = nullptr) { - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = is_dynamic; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output, prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenMulIntdynamicConvertsCorrectly) { - // aten::mul.int(int a, int b) -> (int) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=0]() - %2 : int = prim::Constant[value=1]() - %3 : int = aten::size(%0, %1) - %4 : int = aten::size(%0, %2) - %5 : int = aten::mul(%3, %4) - %6 : Tensor = aten::add(%0, %5, %2) - return (%6))IR"; - baidu::mirana::poros::MulConverter mulconverter; - std::vector input_data; - input_data.push_back(at::zeros({2, 3}, {at::kCUDA}).to(at::ScalarType::Int)); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::zeros({4, 5}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[1].push_back(at::zeros({2, 3}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[2].push_back(at::zeros({2, 3}, {at::kCUDA}).to(at::ScalarType::Int)); - - mul_div_dynamic_test_helper(graph_IR, &mulconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenDivIntdynamicConvertsCorrectly) { - // aten::div.int(int a, int b) -> (float) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=0]() - %2 : int = prim::Constant[value=1]() - %3 : int = aten::size(%0, %1) - %4 : int = aten::size(%0, %2) - %5 : float = aten::div(%3, %4) - %6 : Tensor = aten::add(%0, %5, %2) - return (%6))IR"; - baidu::mirana::poros::DivConverter divconverter; - std::vector input_data; - input_data.push_back(at::zeros({4, 5}, {at::kCUDA})); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::zeros({10, 8}, {at::kCUDA})); - prewarm_data[1].push_back(at::zeros({4, 5}, {at::kCUDA})); - prewarm_data[2].push_back(at::zeros({4, 5}, {at::kCUDA})); - - mul_div_dynamic_test_helper(graph_IR, &divconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenDivNegIntdynamicConvertsCorrectly) { - // aten::div.int(int a, int b) -> (float) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=0]() - %2 : int = prim::Constant[value=1]() - %3 : int = aten::size(%0, %1) - %4 : int = aten::size(%0, %2) - %34 : int = prim::Constant[value=100]() - %35 : int = aten::sub(%3, %34) - %5 : float = aten::div(%35, %4) - %6 : Tensor = aten::add(%0, %5, %2) - return (%6))IR"; - baidu::mirana::poros::DivConverter divconverter; - std::vector input_data; - input_data.push_back(at::zeros({4, 5}, {at::kCUDA})); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::zeros({10, 8}, {at::kCUDA})); - prewarm_data[1].push_back(at::zeros({4, 5}, {at::kCUDA})); - prewarm_data[2].push_back(at::zeros({4, 5}, {at::kCUDA})); - - mul_div_dynamic_test_helper(graph_IR, &divconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenFloordivIntdynamicConvertsCorrectly) { - // aten::floordiv.int(int a, int b) -> (int) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=0]() - %2 : int = prim::Constant[value=1]() - %3 : int = aten::size(%0, %1) - %4 : int = aten::size(%0, %2) - %5 : int = aten::floordiv(%3, %4) - %6 : Tensor = aten::add(%0, %5, %2) - return (%6))IR"; - baidu::mirana::poros::FloordivConverter floordivconverter; - std::vector input_data; - input_data.push_back(at::zeros({10, 4}, {at::kCUDA}).to(at::ScalarType::Int)); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::zeros({12, 5}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[1].push_back(at::zeros({10, 4}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[2].push_back(at::zeros({10, 4}, {at::kCUDA}).to(at::ScalarType::Int)); - - mul_div_dynamic_test_helper(graph_IR, &floordivconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenFloordivNegIntdynamicConvertsCorrectly) { - // aten::floordiv.int(int a, int b) -> (int) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=0]() - %2 : int = prim::Constant[value=1]() - %3 : int = aten::size(%0, %1) - %4 : int = aten::size(%0, %2) - %34 : int = prim::Constant[value=100]() - %35 : int = aten::sub(%3, %34) - %5 : int = aten::floordiv(%35, %4) - %6 : Tensor = aten::add(%0, %5, %2) - return (%6))IR"; - baidu::mirana::poros::FloordivConverter floordivconverter; - std::vector input_data; - input_data.push_back(at::zeros({10, 4}, {at::kCUDA}).to(at::ScalarType::Int)); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::zeros({12, 5}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[1].push_back(at::zeros({10, 4}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[2].push_back(at::zeros({10, 4}, {at::kCUDA}).to(at::ScalarType::Int)); - - mul_div_dynamic_test_helper(graph_IR, &floordivconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenRoundToZeroFloordivIntdynamicConvertsCorrectly) { - // aten::__round_to_zero_floordiv(int a, int b) -> (int) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=0]() - %2 : int = prim::Constant[value=1]() - %3 : int = aten::size(%0, %1) - %4 : int = aten::size(%0, %2) - %5 : int = aten::__round_to_zero_floordiv(%3, %4) - %6 : Tensor = aten::add(%0, %5, %2) - return (%6))IR"; - baidu::mirana::poros::FloordivConverter floordivconverter; - std::vector input_data; - input_data.push_back(at::zeros({10, 4}, {at::kCUDA}).to(at::ScalarType::Int)); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::zeros({12, 5}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[1].push_back(at::zeros({10, 4}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[2].push_back(at::zeros({10, 4}, {at::kCUDA}).to(at::ScalarType::Int)); - - mul_div_dynamic_test_helper(graph_IR, &floordivconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenRoundToZeroFloordivNegIntdynamicConvertsCorrectly) { - // aten::__round_to_zero_floordiv(int a, int b) -> (int) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=0]() - %2 : int = prim::Constant[value=1]() - %3 : int = aten::size(%0, %1) - %4 : int = aten::size(%0, %2) - %34 : int = prim::Constant[value=100]() - %35 : int = aten::sub(%3, %34) - %5 : int = aten::__round_to_zero_floordiv(%35, %4) - %6 : Tensor = aten::add(%0, %5, %2) - return (%6))IR"; - baidu::mirana::poros::FloordivConverter floordivconverter; - std::vector input_data; - input_data.push_back(at::zeros({10, 4}, {at::kCUDA}).to(at::ScalarType::Int)); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::zeros({12, 5}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[1].push_back(at::zeros({10, 4}, {at::kCUDA}).to(at::ScalarType::Int)); - prewarm_data[2].push_back(at::zeros({10, 4}, {at::kCUDA}).to(at::ScalarType::Int)); - - mul_div_dynamic_test_helper(graph_IR, &floordivconverter, input_data, true, &prewarm_data); -} \ No newline at end of file diff --git a/poros/unittest/converter/norm_test.cpp b/poros/unittest/converter/norm_test.cpp deleted file mode 100644 index e0e53a7e505..00000000000 --- a/poros/unittest/converter/norm_test.cpp +++ /dev/null @@ -1,122 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file norm_test.cpp -* @author Lin Xiao Chun (linxiaochun@baidu.com) -* @date 2022-02-23 20:38:15 -* @brief -**/ - -#include -#include - -#include "poros/converter/gpu/norm.h" -#include "poros/util/test_util.h" - -static void norm_test_helper(const std::string &graph_IR, - baidu::mirana::poros::IConverter *converter, - std::vector shape1 = {5}) { - std::vector input_data; - input_data.push_back(at::randn(shape1, {at::kCUDA})); - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output)); - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -static std::string gen_norm_tensor_graph(const std::string &p, const std::string &dims, const std::string &keepdim) { - return R"IR( -graph(%1 : Tensor): - %2 : bool = prim::Constant[value=)IR" + keepdim + R"IR(]() - %3 : int = prim::Constant[value=)IR" + p + R"IR(]() - %4 : int[] = prim::Constant[value=)IR" + dims + R"IR(]() - %5 : Tensor = aten::norm(%1, %3, %4, %2) - return (%5) -)IR"; -} - -static std::string gen_norm_empty_dims_graph(const std::string &p, const std::string &dims, const std::string &keepdim) { - return R"IR( -graph(%1 : Tensor): - %2 : bool = prim::Constant[value=)IR" + keepdim + R"IR(]() - %3 : int = prim::Constant[value=)IR" + p + R"IR(]() - %4 : int[] = prim::ListConstruct() - %5 : Tensor = aten::norm(%1, %3, %4, %2) - return (%5) -)IR"; -} - - - -TEST(Converters, ATenNormConvertsCorrectlyWith) { - std::vector graphIRs; - graphIRs.push_back(gen_norm_tensor_graph("2", "[0]","0")); - graphIRs.push_back(gen_norm_tensor_graph("2", "[1]","0")); - graphIRs.push_back(gen_norm_empty_dims_graph("2", "","0")); - graphIRs.push_back(gen_norm_tensor_graph("2", "[1,2]","0")); - graphIRs.push_back(gen_norm_tensor_graph("2", "[-2,2]","0")); - graphIRs.push_back(gen_norm_tensor_graph("2", "[1,2]","1")); - graphIRs.push_back(gen_norm_tensor_graph("1.5", "[1,2]","0")); - graphIRs.push_back(gen_norm_tensor_graph("0.2", "[-1,-2,-3,-4]","1")); - - baidu::mirana::poros::NormConverter converter; - - for(auto ir:graphIRs){ - norm_test_helper(ir, &converter, {3,4,5,6,7}); - } -} - -static std::string gen_frobenius_norm_tensor_graph(const std::string &dims, const std::string &keepdim) { - return R"IR( -graph(%1 : Tensor): - %2 : bool = prim::Constant[value=)IR" + keepdim + R"IR(]() - %4 : int[] = prim::Constant[value=)IR" + dims + R"IR(]() - %5 : Tensor = aten::frobenius_norm(%1, %4, %2) - return (%5) -)IR"; -} - -static std::string gen_frobenius_norm_empty_dims_graph(const std::string &dims, const std::string &keepdim) { - return R"IR( -graph(%1 : Tensor): - %2 : bool = prim::Constant[value=)IR" + keepdim + R"IR(]() - %4 : int[] = prim::ListConstruct() - %5 : Tensor = aten::frobenius_norm(%1, %4, %2) - return (%5) -)IR"; -} - -TEST(Converters, ATenFrobeniusNormConvertsCorrectlyWith) { - std::vector graphIRs; - graphIRs.push_back(gen_frobenius_norm_tensor_graph( "[0]","0")); - graphIRs.push_back(gen_frobenius_norm_tensor_graph("[1]","0")); - graphIRs.push_back(gen_frobenius_norm_empty_dims_graph("","0")); - graphIRs.push_back(gen_frobenius_norm_tensor_graph("[1,2]","0")); - graphIRs.push_back(gen_frobenius_norm_tensor_graph("[-2,2]","0")); - graphIRs.push_back(gen_frobenius_norm_tensor_graph("[1,2]","1")); - graphIRs.push_back(gen_frobenius_norm_tensor_graph( "[1,2]","0")); - - baidu::mirana::poros::FrobeniusNormConverter converter; - - for(auto ir:graphIRs){ - norm_test_helper(ir, &converter, {3,4,5,6,7}); - } -} \ No newline at end of file diff --git a/poros/unittest/converter/pooling_test.cpp b/poros/unittest/converter/pooling_test.cpp deleted file mode 100644 index 8e780fefe12..00000000000 --- a/poros/unittest/converter/pooling_test.cpp +++ /dev/null @@ -1,268 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file pooling_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/pooling.h" -#include "poros/util/test_util.h" - -static void pooling_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - std::vector shape) { - std::vector input_data; - // input_data.push_back(at::randn(shape, {at::kCUDA})); - input_data.push_back(at::randint(-50, 50, shape, {at::kCUDA})); - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -static std::string gen_maxpool_graph(const std::string& op, - const std::string& kernel_size, - const std::string& stride, - const std::string& padding, - const std::string& dilation, - const std::string& ceil_mode) { - return R"IR( - graph(%0 : Tensor): - %1 : int[] = prim::Constant[value=[)IR" + kernel_size + R"IR(]]() - %2 : int[] = prim::Constant[value=[)IR" + stride + R"IR(]]() - %3 : int[] = prim::Constant[value=[)IR" + padding + R"IR(]]() - %4 : int[] = prim::Constant[value=[)IR" + dilation + R"IR(]]() - %5 : bool = prim::Constant[value=)IR" + ceil_mode + R"IR(]() - %6 : Tensor = aten::)IR" + op + R"IR((%0, %1, %2, %3, %4, %5) - return (%6))IR"; -} - -static std::string gen_avgpool_graph(const std::string& op, - const std::string& kernel_size, - const std::string& stride, - const std::string& padding, - const std::string& ceil_mode, - const std::string& count_include_pad, - const std::string& divisor_override) { - std::string divisor_ir(""); - std::string op_ir(""); - if (divisor_override.empty()) { - divisor_ir = "None = prim::Constant()"; - } else { - divisor_ir = "int = prim::Constant[value=" + divisor_override + "]()"; - } - if (op == "avg_pool1d") { - op_ir = op + "(%0, %1, %2, %3, %4, %5)"; - } else { - op_ir = op + "(%0, %1, %2, %3, %4, %5, %6)"; - } - return R"IR( - graph(%0 : Tensor): - %1 : int[] = prim::Constant[value=[)IR" + kernel_size + R"IR(]]() - %2 : int[] = prim::Constant[value=[)IR" + stride + R"IR(]]() - %3 : int[] = prim::Constant[value=[)IR" + padding + R"IR(]]() - %4 : bool = prim::Constant[value=)IR" + ceil_mode + R"IR(]() - %5 : bool = prim::Constant[value=)IR" + count_include_pad + R"IR(]() - %6 : )IR" + divisor_ir + R"IR( - %7 : Tensor = aten::)IR" + op_ir + R"IR( - return (%7))IR"; -} - -TEST(Converters, ATenMaxPool1DConvertsCorrectly) { - // aten::max_pool1d(Tensor self, int[1] kernel_size, int[1] stride=[], int[1] padding=0, int[1] dilation=1, bool ceil_mode=False) -> Tensor - const auto graph_IR = gen_maxpool_graph("max_pool1d", "3", "2", "1", "1", "0"); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 1, 7}); - pooling_test_helper(graph_IR, &poolingconverter, {1, 1, 8}); -} - -TEST(Converters, ATenMaxPool1DCeilConvertsCorrectly) { - // aten::max_pool1d(Tensor self, int[1] kernel_size, int[1] stride=[], int[1] padding=0, int[1] dilation=1, bool ceil_mode=False) -> Tensor - const auto graph_IR = gen_maxpool_graph("max_pool1d", "3", "2", "1", "1", "1"); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 1, 8}); - pooling_test_helper(graph_IR, &poolingconverter, {1, 1, 7}); -} - -TEST(Converters, ATenMaxPool2DConvertsCorrectly) { - // aten::max_pool2d(Tensor self, int[2] kernel_size, int[2] stride=[], int[2] padding=0, int[2] dilation=1, bool ceil_mode=False) -> Tensor - const auto graph_IR = gen_maxpool_graph("max_pool2d", "3, 3", "2, 2", "1, 1", "1, 1", "0"); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 7, 7}); - pooling_test_helper(graph_IR, &poolingconverter, {1, 8, 8}); -} - -TEST(Converters, ATenMaxPool2DCeilConvertsCorrectly) { - // aten::max_pool2d(Tensor self, int[2] kernel_size, int[2] stride=[], int[2] padding=0, int[2] dilation=1, bool ceil_mode=False) -> Tensor - const auto graph_IR = gen_maxpool_graph("max_pool2d", "3, 3", "2, 2", "1, 1", "1, 1", "1"); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 8, 8}); - pooling_test_helper(graph_IR, &poolingconverter, {1, 7, 7}); -} - -TEST(Converters, ATenMaxPool3DConvertsCorrectly) { - // aten::max_pool3d(Tensor self, int[3] kernel_size, int[3] stride=[], int[3] padding=0, int[3] dilation=1, bool ceil_mode=False) -> Tensor - const auto graph_IR = gen_maxpool_graph("max_pool3d", "3, 3, 3", "2, 2, 2", "1, 1, 1", "1, 1, 1", "0"); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 3, 7, 7, 7}); - pooling_test_helper(graph_IR, &poolingconverter, {1, 3, 8, 8, 8}); -} - -TEST(Converters, ATenMaxPool3DCeilConvertsCorrectly) { - // aten::max_pool3d(Tensor self, int[3] kernel_size, int[3] stride=[], int[3] padding=0, int[3] dilation=1, bool ceil_mode=False) -> Tensor - const auto graph_IR = gen_maxpool_graph("max_pool3d", "3, 3, 3", "2, 2, 2", "1, 1, 1", "1, 1, 1", "1"); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 3, 8, 8, 8}); - pooling_test_helper(graph_IR, &poolingconverter, {1, 3, 7, 7, 7}); -} - -TEST(Converters, ATenAvgPool1DConvertsCorrectly) { - // aten::avg_pool1d(Tensor self, int[1] kernel_size, int[1] stride=[], int[1] padding=0, bool ceil_mode=False, bool count_include_pad=True) -> Tensor - const auto graph_IR = gen_avgpool_graph("avg_pool1d", "3", "2", "1", "0", "1", ""); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 1, 7}); - pooling_test_helper(graph_IR, &poolingconverter, {1, 1, 8}); -} - -TEST(Converters, ATenAvgPool1DCeilConvertsCorrectly) { - // aten::avg_pool1d(Tensor self, int[1] kernel_size, int[1] stride=[], int[1] padding=0, bool ceil_mode=False, bool count_include_pad=True) -> Tensor - const auto graph_IR = gen_avgpool_graph("avg_pool1d", "3", "2", "1", "1", "1", ""); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 1, 7}); - // pooling_test_helper(graph_IR, &poolingconverter, {1, 1, 8}); // fail -} - -TEST(Converters, ATenAvgPool1DNoCountPadConvertsCorrectly) { - // aten::avg_pool1d(Tensor self, int[1] kernel_size, int[1] stride=[], int[1] padding=0, bool ceil_mode=False, bool count_include_pad=True) -> Tensor - const auto graph_IR = gen_avgpool_graph("avg_pool1d", "3", "2", "1", "0", "0", ""); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 1, 7}); - pooling_test_helper(graph_IR, &poolingconverter, {1, 1, 8}); -} - -TEST(Converters, ATenAvgPool1DCeilNoCountPadConvertsCorrectly) { - // aten::avg_pool1d(Tensor self, int[1] kernel_size, int[1] stride=[], int[1] padding=0, bool ceil_mode=False, bool count_include_pad=True) -> Tensor - const auto graph_IR = gen_avgpool_graph("avg_pool1d", "3", "2", "1", "1", "0", ""); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 1, 7}); - pooling_test_helper(graph_IR, &poolingconverter, {1, 1, 8}); -} - -TEST(Converters, ATenAvgPool2DConvertsCorrectly) { - // aten::avg_pool2d(Tensor self, int[2] kernel_size, int[2] stride=[], int[2] padding=0, bool ceil_mode=False, bool count_include_pad=True, int? divisor_override=None) -> Tensor - const auto graph_IR = gen_avgpool_graph("avg_pool2d", "3, 3", "2, 2", "1, 1", "0", "1", ""); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 7, 7}); - pooling_test_helper(graph_IR, &poolingconverter, {1, 8, 8}); -} - -TEST(Converters, ATenAvgPool2DCeilConvertsCorrectly) { - // aten::avg_pool2d(Tensor self, int[2] kernel_size, int[2] stride=[], int[2] padding=0, bool ceil_mode=False, bool count_include_pad=True, int? divisor_override=None) -> Tensor - const auto graph_IR = gen_avgpool_graph("avg_pool2d", "3, 3", "2, 2", "1, 1", "1", "1", ""); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 7, 7}); - // pooling_test_helper(graph_IR, &poolingconverter, {1, 8, 8}); // fail -} - -TEST(Converters, ATenAvgPool2DNoCountPadConvertsCorrectly) { - // aten::avg_pool2d(Tensor self, int[2] kernel_size, int[2] stride=[], int[2] padding=0, bool ceil_mode=False, bool count_include_pad=True, int? divisor_override=None) -> Tensor - const auto graph_IR = gen_avgpool_graph("avg_pool2d", "3, 3", "2, 2", "1, 1", "0", "0", ""); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 7, 7}); - pooling_test_helper(graph_IR, &poolingconverter, {1, 8, 8}); -} - -TEST(Converters, ATenAvgPool2DCeilNoCountPadConvertsCorrectly) { - // aten::avg_pool2d(Tensor self, int[2] kernel_size, int[2] stride=[], int[2] padding=0, bool ceil_mode=False, bool count_include_pad=True, int? divisor_override=None) -> Tensor - const auto graph_IR = gen_avgpool_graph("avg_pool2d", "3, 3", "2, 2", "1, 1", "1", "0", ""); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 7, 7}); - pooling_test_helper(graph_IR, &poolingconverter, {1, 8, 8}); -} - -TEST(Converters, ATenAvgPool2DDivConvertsCorrectly) { - // aten::avg_pool2d(Tensor self, int[2] kernel_size, int[2] stride=[], int[2] padding=0, bool ceil_mode=False, bool count_include_pad=True, int? divisor_override=None) -> Tensor - const auto graph_IR = gen_avgpool_graph("avg_pool2d", "3, 3", "2, 2", "1, 1", "0", "1", "4"); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 7, 7}); - pooling_test_helper(graph_IR, &poolingconverter, {1, 8, 8}); -} - -TEST(Converters, ATenAvgPool2DNegtiveDivConvertsCorrectly) { - // aten::avg_pool2d(Tensor self, int[2] kernel_size, int[2] stride=[], int[2] padding=0, bool ceil_mode=False, bool count_include_pad=True, int? divisor_override=None) -> Tensor - const auto graph_IR = gen_avgpool_graph("avg_pool2d", "3, 3", "2, 2", "1, 1", "0", "1", "-4"); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 7, 7}); - pooling_test_helper(graph_IR, &poolingconverter, {1, 8, 8}); -} - - -TEST(Converters, ATenAvgPool3DConvertsCorrectly) { - // aten::avg_pool3d(Tensor self, int[3] kernel_size, int[3] stride=[], int[3] padding=0, bool ceil_mode=False, bool count_include_pad=True, int? divisor_override=None) -> Tensor - const auto graph_IR = gen_avgpool_graph("avg_pool3d", "3, 3, 3", "2, 2, 2", "1, 1, 1", "0", "1", ""); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 3, 7, 7, 7}); - pooling_test_helper(graph_IR, &poolingconverter, {1, 3, 8, 8, 8}); -} - -TEST(Converters, ATenAvgPool3DCeilConvertsCorrectly) { - // aten::avg_pool3d(Tensor self, int[3] kernel_size, int[3] stride=[], int[3] padding=0, bool ceil_mode=False, bool count_include_pad=True, int? divisor_override=None) -> Tensor - const auto graph_IR = gen_avgpool_graph("avg_pool3d", "3, 3, 3", "2, 2, 2", "1, 1, 1", "1", "1", ""); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 3, 7, 7, 7}); - // pooling_test_helper(graph_IR, &poolingconverter, {1, 3, 8, 8, 8}); // fail -} - -TEST(Converters, ATenAvgPool3DNoCountPadConvertsCorrectly) { - // aten::avg_pool3d(Tensor self, int[3] kernel_size, int[3] stride=[], int[3] padding=0, bool ceil_mode=False, bool count_include_pad=True, int? divisor_override=None) -> Tensor - const auto graph_IR = gen_avgpool_graph("avg_pool3d", "3, 3, 3", "2, 2, 2", "1, 1, 1", "0", "0", ""); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 3, 7, 7, 7}); - pooling_test_helper(graph_IR, &poolingconverter, {1, 3, 8, 8, 8}); -} - -TEST(Converters, ATenAvgPool3DCeilNoCountPadConvertsCorrectly) { - // aten::avg_pool3d(Tensor self, int[3] kernel_size, int[3] stride=[], int[3] padding=0, bool ceil_mode=False, bool count_include_pad=True, int? divisor_override=None) -> Tensor - const auto graph_IR = gen_avgpool_graph("avg_pool3d", "3, 3, 3", "2, 2, 2", "1, 1, 1", "1", "0", ""); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 3, 7, 7, 7}); - pooling_test_helper(graph_IR, &poolingconverter, {1, 3, 8, 8, 8}); -} - -TEST(Converters, ATenAvgPool3DDivConvertsCorrectly) { - // aten::avg_pool3d(Tensor self, int[3] kernel_size, int[3] stride=[], int[3] padding=0, bool ceil_mode=False, bool count_include_pad=True, int? divisor_override=None) -> Tensor - const auto graph_IR = gen_avgpool_graph("avg_pool3d", "3, 3, 3", "2, 2, 2", "1, 1, 1", "0", "1", "8"); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 3, 7, 7, 7}); - pooling_test_helper(graph_IR, &poolingconverter, {1, 3, 8, 8, 8}); -} - -TEST(Converters, ATenAvgPool3DNegtiveDivConvertsCorrectly) { - // aten::avg_pool3d(Tensor self, int[3] kernel_size, int[3] stride=[], int[3] padding=0, bool ceil_mode=False, bool count_include_pad=True, int? divisor_override=None) -> Tensor - const auto graph_IR = gen_avgpool_graph("avg_pool3d", "3, 3, 3", "2, 2, 2", "1, 1, 1", "0", "1", "-8"); - baidu::mirana::poros::PoolingConverter poolingconverter; - pooling_test_helper(graph_IR, &poolingconverter, {1, 3, 7, 7, 7}); - pooling_test_helper(graph_IR, &poolingconverter, {1, 3, 8, 8, 8}); -} \ No newline at end of file diff --git a/poros/unittest/converter/reduce_test.cpp b/poros/unittest/converter/reduce_test.cpp deleted file mode 100644 index 076bc528793..00000000000 --- a/poros/unittest/converter/reduce_test.cpp +++ /dev/null @@ -1,451 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file reduce_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/reduce.h" -#include "poros/util/test_util.h" - -static void reduce_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - std::vector shape1, - bool single_input = true, - std::vector shape2 = {4, 4}, - bool single_output = true, - bool int_flag = false){ - std::vector input_data; - - if(int_flag) { - auto options_pyt_long = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kLong); - input_data.push_back(at::randint(1000, shape1, options_pyt_long)); - } else { - input_data.push_back(at::randn(shape1, {at::kCUDA})); - } - - if (!single_input){ - input_data.push_back(at::randn(shape2, {at::kCUDA})); - } - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output)); - - if (single_output) { - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - } else { - ASSERT_EQ(2, graph_output.size()); - ASSERT_EQ(2, poros_output.size()); - } - - for (size_t i = 0; i < graph_output.size(); i++) { - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[i], poros_output[i], 2e-6)); - } -} - -static std::string gen_basic_graph(const std::string& op) { - return R"IR( - graph(%0 : Tensor): - %1 : None = prim::Constant() - %2 : Tensor = aten::)IR" + - op + R"IR((%0, %1) - return (%2))IR"; -} - -static std::string gen_min_max_graph(const std::string& op) { - return R"IR( - graph(%0 : Tensor): - %1 : Tensor = aten::)IR" + - op + R"IR((%0) - return (%1))IR"; -} - -static std::string gen_min_max_other_graph(const std::string& op) { - return R"IR( - graph(%0 : Tensor, %1 : Tensor): - %1 : Tensor = aten::)IR" + - op + R"IR((%0, %1) - return (%1))IR"; -} - -static std::string gen_min_max_dim_graph(const std::string& op, const std::string& dim) { - return R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=)IR" + dim + R"IR(]() - %2 : bool = prim::Constant[value=0]() - %3 : Tensor, %4 : Tensor = aten::)IR" + op + R"IR((%0, %1, %2) - return (%3, %4))IR"; -} - -static std::string gen_argmin_argmax_graph(const std::string& op, const std::string& dim, const std::string& keepdim) { - return R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=)IR" + dim + R"IR(]() - %2 : bool = prim::Constant[value=)IR" + keepdim + R"IR(]() - %3 : Tensor = aten::)IR" + op + R"IR((%0, %1, %2) - return (%3))IR"; -} - -static std::string gen_argmin_argmax_dim_none_graph(const std::string& op, const std::string& keepdim) { - return R"IR( - graph(%0 : Tensor): - %1 : None = prim::Constant() - %2 : bool = prim::Constant[value=)IR" + keepdim + R"IR(]() - %3 : Tensor = aten::)IR" + op + R"IR((%0, %1, %2) - return (%3))IR"; -} - -static std::string gen_mean_sum_dim_graph(const std::string& op, const std::string& dim, const std::string& keepdim) { - return R"IR( - graph(%0 : Tensor): - %1 : int[] = prim::Constant[value=[)IR" + dim + R"IR(]]() - %2 : bool = prim::Constant[value=)IR" + keepdim + R"IR(]() - %3 : None = prim::Constant() - %4 : Tensor = aten::)IR" + op + R"IR((%0, %1, %2, %3) - return (%4))IR"; -} - -static std::string gen_prod_dim_graph(const std::string& op, const std::string& dim, const std::string& keepdim) { - return R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=)IR" + dim + R"IR(]() - %2 : bool = prim::Constant[value=)IR" + keepdim + R"IR(]() - %3 : None = prim::Constant() - %4 : Tensor = aten::)IR" + op + R"IR((%0, %1, %2, %3) - return (%4))IR"; -} - -TEST(Converters, ATenMeanConvertsCorrectly) { - // aten::mean(Tensor self, *, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_basic_graph("mean"); - baidu::mirana::poros::MeanConverter meanconverter; - reduce_test_helper(graph_IR, &meanconverter, {4, 4}); -} - -TEST(Converters, ATenMeanDimConvertsCorrectly) { - // aten::mean.dim(Tensor self, int[1] dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_mean_sum_dim_graph("mean", "1", "0"); - baidu::mirana::poros::MeanConverter meanconverter; - reduce_test_helper(graph_IR, &meanconverter, {4, 4, 4}); -} - -TEST(Converters, ATenMeanMltiDimsConvertsCorrectly) { - // aten::mean.dim(Tensor self, int[1] dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_mean_sum_dim_graph("mean", "0, 1", "0"); - baidu::mirana::poros::MeanConverter meanconverter; - reduce_test_helper(graph_IR, &meanconverter, {4, 4, 4}); -} - -TEST(Converters, ATenMeanKeepDimsConvertsCorrectly) { - // aten::mean.dim(Tensor self, int[1] dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_mean_sum_dim_graph("mean", "1", "1"); - baidu::mirana::poros::MeanConverter meanconverter; - reduce_test_helper(graph_IR, &meanconverter, {4, 4}); -} - -TEST(Converters, ATenMeanDimNegOneIndexConvertsCorrectly) { - // aten::mean.dim(Tensor self, int[1] dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_mean_sum_dim_graph("mean", "-1", "0"); - baidu::mirana::poros::MeanConverter meanconverter; - reduce_test_helper(graph_IR, &meanconverter, {4, 4, 4}); -} - -TEST(Converters, ATenMeanDimNegOneIndexKeepDimsConvertsCorrectly) { - // aten::mean.dim(Tensor self, int[1] dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_mean_sum_dim_graph("mean", "-1", "1"); - baidu::mirana::poros::MeanConverter meanconverter; - reduce_test_helper(graph_IR, &meanconverter, {4, 4, 4}); -} - -TEST(Converters, ATenMeanDimNegIndexConvertsCorrectly) { - // aten::mean.dim(Tensor self, int[1] dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_mean_sum_dim_graph("mean", "-2", "0"); - baidu::mirana::poros::MeanConverter meanconverter; - reduce_test_helper(graph_IR, &meanconverter, {4, 4, 4}); -} - -TEST(Converters, ATenMeanDimNegIndexKeepDimsConvertsCorrectly) { - // aten::mean.dim(Tensor self, int[1] dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_mean_sum_dim_graph("mean", "-2", "1"); - baidu::mirana::poros::MeanConverter meanconverter; - reduce_test_helper(graph_IR, &meanconverter, {4, 4, 4}); -} - -TEST(Converters, ATenSumConvertsCorrectly) { - // aten::sum(Tensor self, *, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_basic_graph("sum"); - baidu::mirana::poros::SumConverter sumconverter; - reduce_test_helper(graph_IR, &sumconverter, {4, 4}); -} - -TEST(Converters, ATenSumDimConvertsCorrectly) { - // aten::sum.dim_IntList(Tensor self, int[1] dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_mean_sum_dim_graph("sum", "1", "0"); - baidu::mirana::poros::SumConverter sumconverter; - reduce_test_helper(graph_IR, &sumconverter, {4, 4, 4}); -} - -TEST(Converters, ATenSumMltiDimsConvertsCorrectly) { - // aten::sum.dim_IntList(Tensor self, int[1] dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_mean_sum_dim_graph("sum", "0, 1", "0"); - baidu::mirana::poros::SumConverter sumconverter; - reduce_test_helper(graph_IR, &sumconverter, {4, 4, 4}); -} - -TEST(Converters, ATenSumKeepDimsConvertsCorrectly) { - // aten::sum.dim_IntList(Tensor self, int[1] dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_mean_sum_dim_graph("sum", "1", "1"); - baidu::mirana::poros::SumConverter sumconverter; - reduce_test_helper(graph_IR, &sumconverter, {4, 4}); -} - -TEST(Converters, ATenSumDimNegOneIndexConvertsCorrectly) { - // aten::sum.dim_IntList(Tensor self, int[1] dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_mean_sum_dim_graph("sum", "-1", "0"); - baidu::mirana::poros::SumConverter sumconverter; - reduce_test_helper(graph_IR, &sumconverter, {4, 4, 4}); -} - -TEST(Converters, ATenSumDimNegOneIndexKeepDimsConvertsCorrectly) { - // aten::sum.dim_IntList(Tensor self, int[1] dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_mean_sum_dim_graph("sum", "-1", "1"); - baidu::mirana::poros::SumConverter sumconverter; - reduce_test_helper(graph_IR, &sumconverter, {4, 4, 4}); -} - -TEST(Converters, ATenSumDimNegIndexConvertsCorrectly) { - // aten::sum.dim_IntList(Tensor self, int[1] dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_mean_sum_dim_graph("sum", "-2", "0"); - baidu::mirana::poros::SumConverter sumconverter; - reduce_test_helper(graph_IR, &sumconverter, {4, 4, 4}); -} - -TEST(Converters, ATenSumDimNegIndexKeepDimsConvertsCorrectly) { - // aten::sum.dim_IntList(Tensor self, int[1] dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_mean_sum_dim_graph("sum", "-2", "1"); - baidu::mirana::poros::SumConverter sumconverter; - reduce_test_helper(graph_IR, &sumconverter, {4, 4, 4}); -} - -TEST(Converters, ATenProdConvertsCorrectly) { - // aten::prod(Tensor self, *, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_basic_graph("prod"); - baidu::mirana::poros::ProdConverter prodconverter; - reduce_test_helper(graph_IR, &prodconverter, {4, 4}); -} - -TEST(Converters, ATenProdDimConvertsCorrectly) { - // aten::prod.dim_int(Tensor self, int dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_prod_dim_graph("prod", "1", "0"); - baidu::mirana::poros::ProdConverter prodconverter; - reduce_test_helper(graph_IR, &prodconverter, {4, 4, 4}); -} - -TEST(Converters, ATenProdKeepDimsConvertsCorrectly) { - // aten::prod.dim_int(Tensor self, int dim, bool keepdim=False, *, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_prod_dim_graph("prod", "1", "1"); - baidu::mirana::poros::ProdConverter prodconverter; - reduce_test_helper(graph_IR, &prodconverter, {4, 4}); -} - -TEST(Converters, ATenMaxConvertsCorrectly) { - // aten::max(Tensor self) -> Tensor - const auto graph_IR = gen_min_max_graph("max"); - baidu::mirana::poros::MaxMinConverter maxminconverter; - reduce_test_helper(graph_IR, &maxminconverter, {4, 4}); -} - -TEST(Converters, ATenMinConvertsCorrectly) { - // aten::min(Tensor self) -> Tensor - const auto graph_IR = gen_min_max_graph("min"); - baidu::mirana::poros::MaxMinConverter maxminconverter; - reduce_test_helper(graph_IR, &maxminconverter, {4, 4}); -} - -TEST(Converters, ATenMaxOtherConvertsCorrectly) { - // aten::max.other(Tensor self, Tensor other) -> Tensor - const auto graph_IR = gen_min_max_other_graph("max"); - baidu::mirana::poros::MaxMinConverter maxminconverter; - reduce_test_helper(graph_IR, &maxminconverter, {4, 4}, false, {4, 4}); - reduce_test_helper(graph_IR, &maxminconverter, {3, 4}, false, {4}); - reduce_test_helper(graph_IR, &maxminconverter, {4}, false, {3, 4}); - reduce_test_helper(graph_IR, &maxminconverter, {4, 1}, false, {1, 4}); - reduce_test_helper(graph_IR, &maxminconverter, {3, 4, 3}, false, {4, 3}); - reduce_test_helper(graph_IR, &maxminconverter, {4, 3}, false, {3, 4, 3}); -} - -TEST(Converters, ATenMinOtherConvertsCorrectly) { - // aten::min.other(Tensor self, Tensor other) -> Tensor - const auto graph_IR = gen_min_max_other_graph("min"); - baidu::mirana::poros::MaxMinConverter maxminconverter; - reduce_test_helper(graph_IR, &maxminconverter, {4, 4}, false, {4, 4}); - reduce_test_helper(graph_IR, &maxminconverter, {3, 4}, false, {4}); - reduce_test_helper(graph_IR, &maxminconverter, {4}, false, {3, 4}); - reduce_test_helper(graph_IR, &maxminconverter, {4, 1}, false, {1, 4}); - reduce_test_helper(graph_IR, &maxminconverter, {3, 4, 3}, false, {4, 3}); - reduce_test_helper(graph_IR, &maxminconverter, {4, 3}, false, {3, 4, 3}); -} - -TEST(Converters, ATenMaxDimConvertsCorrectly) { - // aten::max.dim(Tensor self, int dim, bool keepdim=False) -> (Tensor values, Tensor indices) - const auto graph_IR = gen_min_max_dim_graph("max", "0"); - baidu::mirana::poros::MaxMinConverter maxminconverter; - reduce_test_helper(graph_IR, &maxminconverter, {4, 5, 3}, true, {}, false); - const auto graph_IR2 = gen_min_max_dim_graph("max", "1"); - reduce_test_helper(graph_IR2, &maxminconverter, {4, 5, 3}, true, {}, false); - const auto graph_IR3 = gen_min_max_dim_graph("max", "-1"); - reduce_test_helper(graph_IR3, &maxminconverter, {4, 5, 3}, true, {}, false); - const auto graph_IR4 = gen_min_max_dim_graph("max", "-1"); - reduce_test_helper(graph_IR4, &maxminconverter, {4, 3}, true, {}, false); -} - -TEST(Converters, ATenMinDimConvertsCorrectly) { - // aten::min.dim(Tensor self, int dim, bool keepdim=False) -> (Tensor values, Tensor indices) - const auto graph_IR = gen_min_max_dim_graph("min", "0"); - baidu::mirana::poros::MaxMinConverter maxminconverter; - reduce_test_helper(graph_IR, &maxminconverter, {4, 5, 3}, true, {}, false); - const auto graph_IR2 = gen_min_max_dim_graph("min", "1"); - reduce_test_helper(graph_IR2, &maxminconverter, {4, 5, 3}, true, {}, false); - const auto graph_IR3 = gen_min_max_dim_graph("min", "-1"); - reduce_test_helper(graph_IR3, &maxminconverter, {4, 5, 3}, true, {}, false); - const auto graph_IR4 = gen_min_max_dim_graph("min", "-1"); - reduce_test_helper(graph_IR4, &maxminconverter, {4, 3}, true, {}, false); -} - -TEST(Converters, ATenMaxDimDynamicConvertsCorrectly) { - // aten::max.dim(Tensor self, int dim, bool keepdim=False) -> (Tensor values, Tensor indices) - const auto graph_IR = gen_min_max_dim_graph("max", "0"); - baidu::mirana::poros::MaxMinConverter maxminconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({4, 5, 6}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({3, 4, 5}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({3, 4, 5}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({3, 4, 5}, {at::kCUDA})); - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = true; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &maxminconverter, - input_data, graph_output, poros_output, &prewarm_data)); - ASSERT_EQ(2, graph_output.size()); - ASSERT_EQ(2, poros_output.size()); - - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[1], poros_output[1], 2e-6)); -} - -TEST(Converters, ATenMinDimDynamicConvertsCorrectly) { - // aten::max.dim(Tensor self, int dim, bool keepdim=False) -> (Tensor values, Tensor indices) - const auto graph_IR = gen_min_max_dim_graph("min", "1"); - baidu::mirana::poros::MaxMinConverter maxminconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({4, 5, 6}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({3, 4, 5}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({3, 4, 5}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({3, 4, 5}, {at::kCUDA})); - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = true; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &maxminconverter, - input_data, graph_output, poros_output, &prewarm_data)); - ASSERT_EQ(2, graph_output.size()); - ASSERT_EQ(2, poros_output.size()); - - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[1], poros_output[1], 2e-6)); -} - -TEST(Converters, ArgmaxConvertersCorrectly) { - // aten::argmax(Tensor self, int? dim=None, bool keepdim=False) -> (Tensor) - baidu::mirana::poros::ArgmaxArgminConverter argmaxargminconverter; - - const auto graph_IR1 = gen_argmin_argmax_graph("argmax", "0", "0"); - reduce_test_helper(graph_IR1, &argmaxargminconverter, {4, 4}, true, {}, true); - const auto graph_IR2 = gen_argmin_argmax_graph("argmax", "1", "0"); - reduce_test_helper(graph_IR2, &argmaxargminconverter, {4, 4}, true, {}, true); - const auto graph_IR3 = gen_argmin_argmax_graph("argmax", "2", "0"); - reduce_test_helper(graph_IR3, &argmaxargminconverter, {4, 4, 6}, true, {}, true); - const auto graph_IR4 = gen_argmin_argmax_graph("argmax", "3", "0"); - reduce_test_helper(graph_IR4, &argmaxargminconverter, {4, 4, 6, 8}, true, {}, true); - - const auto graph_IR5 = gen_argmin_argmax_graph("argmax", "0", "1"); - reduce_test_helper(graph_IR5, &argmaxargminconverter, {4, 4}, true, {}, true); - const auto graph_IR6 = gen_argmin_argmax_graph("argmax", "1", "1"); - reduce_test_helper(graph_IR6, &argmaxargminconverter, {4, 4}, true, {}, true); - const auto graph_IR7 = gen_argmin_argmax_graph("argmax", "-1", "1"); - reduce_test_helper(graph_IR7, &argmaxargminconverter, {4, 4}, true, {}, true); - const auto graph_IR8 = gen_argmin_argmax_graph("argmax", "-1", "0"); - reduce_test_helper(graph_IR8, &argmaxargminconverter, {4, 4}, true, {}, true); - - // test input tensor of int type - const auto graph_IR9 = gen_argmin_argmax_graph("argmax", "1", "0"); - reduce_test_helper(graph_IR9, &argmaxargminconverter, {4, 4}, true, {}, true, true); - const auto graph_IR10 = gen_argmin_argmax_graph("argmax", "-1", "0"); - reduce_test_helper(graph_IR10, &argmaxargminconverter, {4, 4}, true, {}, true, true); -} - -TEST(Converters, ArgminConvertersCorrectly) { - // aten::argmin(Tensor self, int? dim=None, bool keepdim=False) -> (Tensor) - baidu::mirana::poros::ArgmaxArgminConverter argmaxargminconverter; - - const auto graph_IR1 = gen_argmin_argmax_graph("argmin", "0", "0"); - reduce_test_helper(graph_IR1, &argmaxargminconverter, {4, 4}, true, {}, true); - const auto graph_IR2 = gen_argmin_argmax_graph("argmin", "1", "0"); - reduce_test_helper(graph_IR2, &argmaxargminconverter, {4, 4}, true, {}, true); - const auto graph_IR3 = gen_argmin_argmax_graph("argmin", "2", "0"); - reduce_test_helper(graph_IR3, &argmaxargminconverter, {4, 4, 6}, true, {}, true); - const auto graph_IR4 = gen_argmin_argmax_graph("argmin", "3", "0"); - reduce_test_helper(graph_IR4, &argmaxargminconverter, {4, 4, 6, 8}, true, {}, true); - - const auto graph_IR5 = gen_argmin_argmax_graph("argmin", "0", "1"); - reduce_test_helper(graph_IR5, &argmaxargminconverter, {4, 4}, true, {}, true); - const auto graph_IR6 = gen_argmin_argmax_graph("argmin", "1", "1"); - reduce_test_helper(graph_IR6, &argmaxargminconverter, {4, 4}, true, {}, true); - const auto graph_IR7 = gen_argmin_argmax_graph("argmin", "-1", "1"); - reduce_test_helper(graph_IR7, &argmaxargminconverter, {4, 4}, true, {}, true); - - // test input tensor of int type - const auto graph_IR9 = gen_argmin_argmax_graph("argmin", "1", "0"); - reduce_test_helper(graph_IR9, &argmaxargminconverter, {4, 4}, true, {}, true, true); - const auto graph_IR10 = gen_argmin_argmax_graph("argmin", "-1", "0"); - reduce_test_helper(graph_IR10, &argmaxargminconverter, {4, 4}, true, {}, true, true); -} - -// TODO: to imp dim=None -// TEST(Converters, ArgmaxNoneDimConvertersCorrectly) { -// // aten::argmax(Tensor self, int? dim=None, bool keepdim=False) -> (Tensor) -// baidu::mirana::poros::ArgmaxArgminConverter argmaxargminconverter; -// const auto graph_IR1 = gen_argmin_argmax_dim_none_graph("argmax", "0"); -// reduce_test_helper(graph_IR1, &argmaxargminconverter, {4, 4}, true, {}, true); -// } \ No newline at end of file diff --git a/poros/unittest/converter/reflection_pad_test.cpp b/poros/unittest/converter/reflection_pad_test.cpp deleted file mode 100644 index 8408e965191..00000000000 --- a/poros/unittest/converter/reflection_pad_test.cpp +++ /dev/null @@ -1,137 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file reflection_pad_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/reflection_pad.h" -#include "poros/util/test_util.h" - -static void reflection_pad_test_helper(const std::string& graph_IR, - std::vector shape, - bool is_dynamic = false, - std::vector>* prewarm_data = nullptr) { - std::vector input_data; - input_data.push_back(at::randn(shape, {at::kCUDA})); - - baidu::mirana::poros::ReflectionPadConverter reflectionpadconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = is_dynamic; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &reflectionpadconverter, - input_data, graph_output, poros_output, prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -static std::string gen_reflection_pad_graph(const std::string& op, - const std::string& padding) { - return R"IR( - graph(%0 : Tensor): - %1 : int[] = prim::Constant[value=[)IR" + padding + R"IR(]]() - %2 : Tensor = aten::)IR" + op + R"IR((%0, %1) - return (%2))IR"; -} - -TEST(Converters, ATenReflectionPad1DConvertsCorrectly) { - // aten::reflection_pad1d(Tensor self, int[2] padding) -> Tensor - const auto graph_IR = gen_reflection_pad_graph("reflection_pad1d", "2, 2"); - reflection_pad_test_helper(graph_IR, {2, 5}); -} - -TEST(Converters, ATenReflectionPad2DConvertsCorrectly) { - // aten::reflection_pad2d(Tensor self, int[4] padding) -> Tensor - const auto graph_IR = gen_reflection_pad_graph("reflection_pad2d", "1, 1, 2, 3"); - reflection_pad_test_helper(graph_IR, {3, 4, 3}); -} - -TEST(Converters, ATenReflectionPad1DDynamicConvertsCorrectly) { - // aten::reflection_pad1d(Tensor self, int[2] padding) -> Tensor - const auto graph_IR = gen_reflection_pad_graph("reflection_pad1d", "2, 3"); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({3, 6}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({2, 5}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({2, 5}, {at::kCUDA})); - - reflection_pad_test_helper(graph_IR, {2, 5}, true, &prewarm_data); -} - -TEST(Converters, ATenReflectionPad2DDynamicConvertsCorrectly) { - // aten::reflection_pad2d(Tensor self, int[4] padding) -> Tensor - const auto graph_IR = gen_reflection_pad_graph("reflection_pad2d", "1, 1, 2, 3"); - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({4, 5, 4}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({3, 4, 3}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({3, 4, 3}, {at::kCUDA})); - - reflection_pad_test_helper(graph_IR, {3, 4, 3}, true, &prewarm_data); -} - -TEST(Converters, ATenReflectionPad1DDynamicscalarinputConvertsCorrectly) { - // aten::reflection_pad2d(Tensor self, int[4] padding) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=1]() - %2 : int = prim::Constant[value=1]() - %3 : int = prim::Constant[value=2]() - %4 : int = aten::size(%0, %1) - %5 : float = aten::div(%4, %3) - %6 : int = aten::floor(%5) - %7 : int[] = prim::ListConstruct(%1, %6) - %8 : Tensor = aten::reflection_pad1d(%0, %7) - return (%8))IR"; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({3, 7}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({2, 5}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({2, 5}, {at::kCUDA})); - - reflection_pad_test_helper(graph_IR, {2, 7}, true, &prewarm_data); -} - - -TEST(Converters, ATenReflectionPad2DDynamicscalarinputConvertsCorrectly) { - // aten::reflection_pad2d(Tensor self, int[4] padding) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=1]() - %2 : int = prim::Constant[value=1]() - %3 : int = prim::Constant[value=2]() - %4 : int = aten::size(%0, %1) - %5 : float = aten::div(%4, %3) - %6 : int = aten::floor(%5) - %7 : int[] = prim::ListConstruct(%1, %2, %3, %6) - %8 : Tensor = aten::reflection_pad2d(%0, %7) - return (%8))IR"; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({4, 7, 4}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({3, 5, 3}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({3, 5, 3}, {at::kCUDA})); - - reflection_pad_test_helper(graph_IR, {3, 5, 3}, true, &prewarm_data); - -} \ No newline at end of file diff --git a/poros/unittest/converter/replication_pad_test.cpp b/poros/unittest/converter/replication_pad_test.cpp deleted file mode 100644 index 4e0d331e6a8..00000000000 --- a/poros/unittest/converter/replication_pad_test.cpp +++ /dev/null @@ -1,105 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file replication_pad_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/replication_pad.h" -#include "poros/util/test_util.h" - -static void replicationpad_test_helper(const std::string& graph_IR, - std::vector shape) { - std::vector input_data; - input_data.push_back(at::randn(shape, {at::kCUDA})); - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - baidu::mirana::poros::ReplicationPadConverter replicationpadconverter; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &replicationpadconverter, - input_data, graph_output, poros_output)); - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - // ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -static std::string gen_replicationpad_graph(const std::string& op, - const std::string& padding) { - return R"IR( - graph(%0 : Tensor): - %1 : int[] = prim::Constant[value=[)IR" + padding + R"IR(]]() - %2 : Tensor = aten::)IR" + op + R"IR((%0, %1) - return (%2))IR"; -} - -TEST(Converters, ATenReplicationPad1DConvertsCorrectly) { - // aten::replication_pad1d(Tensor self, int[2] padding) -> Tensor - const auto graph_IR = gen_replicationpad_graph("replication_pad1d", "2, 3"); - replicationpad_test_helper(graph_IR, {1, 3, 4}); -} - -TEST(Converters, ATenReplicationPad1DRightZeroConvertsCorrectly) { - // aten::replication_pad1d(Tensor self, int[2] padding) -> Tensor - const auto graph_IR = gen_replicationpad_graph("replication_pad1d", "2, 0"); - replicationpad_test_helper(graph_IR, {1, 3, 4}); -} - -TEST(Converters, ATenReplicationPad1DLeftZeroConvertsCorrectly) { - // aten::replication_pad1d(Tensor self, int[2] padding) -> Tensor - const auto graph_IR = gen_replicationpad_graph("replication_pad1d", "0, 3"); - replicationpad_test_helper(graph_IR, {1, 3, 4}); -} - -TEST(Converters, ATenReplicationPad2DConvertsCorrectly) { - // aten::replication_pad2d(Tensor self, int[4] padding) -> Tensor - const auto graph_IR = gen_replicationpad_graph("replication_pad2d", "2, 3, 2, 3"); - replicationpad_test_helper(graph_IR, {1, 3, 4, 5}); -} - -TEST(Converters, ATenReplicationPad2DBottomZeroConvertsCorrectly) { - // aten::replication_pad2d(Tensor self, int[4] padding) -> Tensor - const auto graph_IR = gen_replicationpad_graph("replication_pad2d", "2, 0, 2, 0"); - replicationpad_test_helper(graph_IR, {1, 3, 4, 5}); -} - -TEST(Converters, ATenReplicationPad2DTopZeroConvertsCorrectly) { - // aten::replication_pad2d(Tensor self, int[4] padding) -> Tensor - const auto graph_IR = gen_replicationpad_graph("replication_pad2d", "0, 3, 0, 3"); - replicationpad_test_helper(graph_IR, {1, 3, 4, 5}); -} - -TEST(Converters, ATenReplicationPad3DConvertsCorrectly) { - // aten::replication_pad3d(Tensor self, int[6] padding) -> Tensor - const auto graph_IR = gen_replicationpad_graph("replication_pad3d", "2, 3, 2, 3, 1, 4"); - replicationpad_test_helper(graph_IR, {1, 3, 4, 5, 3}); -} - -TEST(Converters, ATenReplicationPad3DRightBottomZeroConvertsCorrectly) { - // aten::replication_pad3d(Tensor self, int[6] padding) -> Tensor - const auto graph_IR = gen_replicationpad_graph("replication_pad3d", "2, 0, 2, 0, 1, 0"); - replicationpad_test_helper(graph_IR, {1, 3, 4, 5, 3}); -} - -TEST(Converters, ATenReplicationPad3DLeftTopZeroConvertsCorrectly) { - // aten::replication_pad3d(Tensor self, int[6] padding) -> Tensor - const auto graph_IR = gen_replicationpad_graph("replication_pad3d", "2, 0, 2, 0, 1, 0"); - replicationpad_test_helper(graph_IR, {1, 3, 4, 5, 3}); -} \ No newline at end of file diff --git a/poros/unittest/converter/roll_test.cpp b/poros/unittest/converter/roll_test.cpp deleted file mode 100644 index 7bb4ee02df0..00000000000 --- a/poros/unittest/converter/roll_test.cpp +++ /dev/null @@ -1,79 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file roll_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Jul 20 19:34:51 CST 2022 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/roll.h" -#include "poros/util/test_util.h" - -static void roll_test_helper(const std::string& graph_IR, - std::vector shape, - bool is_dynamic = false, - std::vector>* prewarm_data = nullptr) { - std::vector input_data; - int64_t shape_mul = 1; - for (int64_t& s : shape) { - shape_mul *= s; - } - input_data.push_back(at::randint(0, shape_mul, shape, {at::kCUDA})); - - baidu::mirana::poros::RollConverter rollconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = is_dynamic; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &rollconverter, - input_data, graph_output, poros_output, prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -static std::string gen_roll_graph(const std::string& shifts, const std::string& dims) { - return R"IR( - graph(%0 : Tensor): - %1 : int[] = prim::Constant[value=)IR" + shifts + R"IR(]() - %2 : int[] = prim::Constant[value=)IR" + dims + R"IR(]() - %3 : Tensor = aten::roll(%0, %1, %2) - return (%3))IR"; -} - -TEST(Converters, ATenRollConvertsCorrectly) { - // aten::roll(Tensor self, int[1] shifts, int[1] dims=[]) -> (Tensor) - const std::string graph_IR = gen_roll_graph("[-1, 0, -2, 3]", "[0, 1, 2, 3]"); - roll_test_helper(graph_IR, {4, 4, 4, 4}); -} - - -TEST(Converters, ATenRollConvertsCorrectlyShiftsGreaterThanDims) { - // aten::roll(Tensor self, int[1] shifts, int[1] dims=[]) -> (Tensor) - const std::string graph_IR = gen_roll_graph("[-99, 100, 51, -21]", "[0, 1, 2, 3]"); - roll_test_helper(graph_IR, {4, 4, 4, 4}); -} - -TEST(Converters, ATenRollConvertsCorrectlyShiftSomeDims) { - // aten::roll(Tensor self, int[1] shifts, int[1] dims=[]) -> (Tensor) - const std::string graph_IR = gen_roll_graph("[0, -2, 3]", "[0, 1, 3]"); - roll_test_helper(graph_IR, {4, 4, 4, 4}); -} \ No newline at end of file diff --git a/poros/unittest/converter/select_test.cpp b/poros/unittest/converter/select_test.cpp deleted file mode 100644 index 8e1ac6ef767..00000000000 --- a/poros/unittest/converter/select_test.cpp +++ /dev/null @@ -1,1350 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file select_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/select.h" -#include "poros/util/test_util.h" - -static void select_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - std::vector shape, - bool is_dynamic = false, - std::vector>* prewarm_data = nullptr) { - std::vector input_data; - input_data.push_back(at::randn(shape, {at::kCUDA})); - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = is_dynamic; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output, prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -static void split_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - std::vector shape, - const int64_t& output_size) { - std::vector input_data; - input_data.push_back(at::randn(shape, {at::kCUDA})); - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(output_size, graph_output.size()); - ASSERT_EQ(output_size, poros_output.size()); - - for (int64_t i = 0; i < output_size; i++) { - ASSERT_TRUE(graph_output[i].equal(poros_output[i])); - } -} - -static void embedding_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter) { - - std::vector input_data; - auto options_pyt = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kInt64); - auto weight = at::randn({10, 4}, {at::kCUDA}); - auto input = at::tensor({2, 3, 4}, options_pyt); - input_data.push_back(weight); - input_data.push_back(input); - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output)); - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -static std::string gen_select_graph(const std::string& dim, const std::string& index) { - return R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=)IR" + dim + R"IR(]() - %2 : int = prim::Constant[value=)IR" + index + R"IR(]() - %3 : Tensor = aten::select(%0, %1, %2) - return (%3))IR"; -} - -static std::string gen_slice_graph(const std::string& dim, - const std::string& start, - const std::string& end, - const std::string& step) { - std::string start_ir, end_ir; - if (start.empty()) { - start_ir = "%2 : None = prim::Constant()"; - } else { - start_ir = "%2 : int = prim::Constant[value=" + start + "]()"; - } - if (end.empty()) { - end_ir = "%3 : None = prim::Constant()"; - } else { - end_ir = "%3 : int = prim::Constant[value=" + end + "]()"; - } - return R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=)IR" + dim + R"IR(]() - )IR" + start_ir + R"IR( - )IR" + end_ir + R"IR( - %4 : int = prim::Constant[value=)IR" + step + R"IR(]() - %5 : Tensor = aten::slice(%0, %1, %2, %3, %4) - return (%5))IR"; -} - -static std::string gen_narrow_graph(const std::string& dim, - const std::string& start, - const std::string& length, - bool singleinput) { - if (singleinput) { - return R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=)IR" + dim + R"IR(]() - %2 : int = prim::Constant[value=)IR" + start + R"IR(]() - %3 : int = prim::Constant[value=)IR" + length + R"IR(]() - %4 : Tensor = aten::narrow(%0, %1, %2, %3) - return (%4))IR"; - } else { - return R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : int = prim::Constant[value=)IR" + dim + R"IR(]() - %3 : int = prim::Constant[value=)IR" + length + R"IR(]() - %4 : Tensor = aten::narrow(%0, %2, %1, %3) - return (%4))IR"; - } -} - -static std::string gen_indexput_graph(const std::string& fold) { - return R"IR( - graph(%x : Tensor): - %none : NoneType = prim::Constant() - %0 : int = prim::Constant[value=0]() - %1 : int = prim::Constant[value=1]() - %2 : int = prim::Constant[value=2]() - %4 : int = prim::Constant[value=4]() - %negtive : int = prim::Constant[value=-1]() - %fold : int = prim::Constant[value=)IR" + fold + R"IR(]() - %false : bool = prim::Constant[value=0]() - - %out : Tensor = aten::zeros_like(%x, %none, %none, %none, %none, %none) - %302 : Tensor = aten::slice(%x, %0, %none, %none, %1) - %303 : Tensor = aten::slice(%302, %1, %1, %none, %1) - %304 : Tensor = aten::slice(%303, %2, %none, %fold, %1) - - %2726 : int = aten::size(%out, %0) - %2731 : Tensor = aten::arange(%2726, %4, %none, %none, %none) - %2733 : Tensor = aten::slice(%2731, %0, %none, %none, %1) - - %2735 : int = aten::size(%out, %1) - %2740 : Tensor = aten::arange(%2735, %4, %none, %none, %none) - %2742 : Tensor = aten::slice(%2740, %0, %none, %negtive, %1) - - %2744 : int = aten::size(%out, %2) - %2749 : Tensor = aten::arange(%2744, %4, %none, %none, %none) - %2751 : Tensor = aten::slice(%2749, %0, %none, %fold, %1) - - %2752 : int[] = prim::Constant[value=[-1, 1, 1]]() - %2753 : Tensor = aten::view(%2733, %2752) - %2754 : int[] = prim::Constant[value=[-1, 1]]() - %2755 : Tensor = aten::view(%2742, %2754) - %2756 : Tensor?[] = prim::ListConstruct(%2753, %2755, %2751) - %2757 : Tensor = aten::index_put(%out, %2756, %304, %false) - return (%2757))IR"; -} - -static std::string gen_indexput_with_singular_value_graph() { - return R"IR( - graph(%x : Tensor): - %false : bool = prim::Constant[value=0]() - %none : NoneType = prim::Constant() - %neg1 : int = prim::Constant[value=-1]() - %0 : int = prim::Constant[value=0]() - %1 : int = prim::Constant[value=1]() - %4 : int = prim::Constant[value=4]() - %device : Device = prim::Constant[value="cuda:0"]() - - %size : int[] = aten::size(%x) - %input_shape : int[] = aten::slice(%size, %none, %neg1, %1) - %attention_mask : Tensor = aten::zeros(%input_shape, %none, %none, %device, %none) - %92 : int = aten::size(%attention_mask, %1) - %90 : Tensor = aten::arange(%92, %4, %none, %none, %none) - %86 : Tensor = aten::slice(%90, %0, %none, %none, %1) - %2326 : int = prim::dtype(%86) - %101 : Tensor = aten::tensor(%0, %2326, %device, %false) - - %index : Tensor?[] = prim::ListConstruct(%101, %86) - - %28 : int = prim::dtype(%attention_mask) - %value : Tensor = aten::tensor(%1, %28, %device, %false) - %tmp : Tensor = aten::index_put(%attention_mask, %index, %value, %false) - %out : Tensor = aten::mul(%tmp, %4) - return (%out))IR"; -} - -TEST(Converters, ATenSelectIntConvertsCorrectly) { - // aten::select.int(Tensor(a) self, int dim, int index) -> Tensor(a) - const auto graph_IR = gen_select_graph("0", "0"); - baidu::mirana::poros::SelectConverter selectconverter; - select_test_helper(graph_IR, &selectconverter, {4, 4, 4}); -} - -TEST(Converters, ATenSelectIntDimIsOneConvertsCorrectly) { - // aten::select.int(Tensor(a) self, int dim, int index) -> Tensor(a) - const auto graph_IR = gen_select_graph("1", "0"); - baidu::mirana::poros::SelectConverter selectconverter; - select_test_helper(graph_IR, &selectconverter, {4, 4, 4}); -} - -TEST(Converters, ATenSelectIntDimNegativeConvertsCorrectly) { - // aten::select.int(Tensor(a) self, int dim, int index) -> Tensor(a) - const auto graph_IR = gen_select_graph("-2", "0"); - baidu::mirana::poros::SelectConverter selectconverter; - select_test_helper(graph_IR, &selectconverter, {4, 4, 4}); -} - -TEST(Converters, ATenSelectIntNegIndexConvertsCorrectly) { - // aten::select.int(Tensor(a) self, int dim, int index) -> Tensor(a) - const auto graph_IR = gen_select_graph("0", "-1"); - baidu::mirana::poros::SelectConverter selectconverter; - select_test_helper(graph_IR, &selectconverter, {4, 4, 4}); -} - -TEST(Converters, ATenSelectSelfDynaimcConverterCorrectly) { - // aten::select.int(Tensor(a) self, int dim, int index) -> Tensor(a) - const auto graph_IR = gen_select_graph("3", "0"); - baidu::mirana::poros::SelectConverter selectconverter; - - std::vector input_data; - auto options_pyt = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kInt); - input_data.push_back(at::randint(0, 100, {3, 4, 5, 6}, options_pyt)); // indices - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randint(0, 3, {3, 4, 5, 6}, options_pyt)); // indices - prewarm_data[1].push_back(at::randint(0, 3, {2, 3, 4, 5}, options_pyt)); // indices - prewarm_data[2].push_back(at::randint(0, 3, {2, 3, 4, 5}, options_pyt)); // indices - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = true; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &selectconverter, - input_data, graph_output, poros_output, &prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenSliceConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = gen_slice_graph("2", "0", "2", "1"); - baidu::mirana::poros::SliceConverter sliceconverter; - select_test_helper(graph_IR, &sliceconverter, {3, 4, 5, 6}); -} - -TEST(Converters, ATenSliceDimNegConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = gen_slice_graph("-2", "0", "2", "1"); - - baidu::mirana::poros::SliceConverter sliceconverter; - select_test_helper(graph_IR, &sliceconverter, {3, 4, 5, 6}); -} - -TEST(Converters, ATenSliceStartNoneConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = gen_slice_graph("2", "", "3", "1"); - baidu::mirana::poros::SliceConverter sliceconverter; - select_test_helper(graph_IR, &sliceconverter, {3, 4, 5, 6}); -} - -TEST(Converters, ATenSliceStartNegConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = gen_slice_graph("2", "-2", "3", "1"); - baidu::mirana::poros::SliceConverter sliceconverter; - select_test_helper(graph_IR, &sliceconverter, {3, 4, 5, 6}); -} - -TEST(Converters, ATenSliceEndNoneConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = gen_slice_graph("2", "1", "", "1"); - baidu::mirana::poros::SliceConverter sliceconverter; - select_test_helper(graph_IR, &sliceconverter, {3, 4, 5, 6}); -} - -TEST(Converters, ATenSliceEndNegConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = gen_slice_graph("2", "0", "-2", "2"); - baidu::mirana::poros::SliceConverter sliceconverter; - select_test_helper(graph_IR, &sliceconverter, {3, 4, 5, 6}); -} - -TEST(Converters, ATenSliceStartEndNegConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = gen_slice_graph("2", "-3", "-1", "2"); - baidu::mirana::poros::SliceConverter sliceconverter; - select_test_helper(graph_IR, &sliceconverter, {3, 4, 5, 6}); -} - -TEST(Converters, ATenSliceStartEndNoneConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = gen_slice_graph("2", "", "", "2"); - baidu::mirana::poros::SliceConverter sliceconverter; - select_test_helper(graph_IR, &sliceconverter, {3, 4, 5, 6}); -} - -TEST(Converters, ATenSliceStepConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = gen_slice_graph("2", "0", "3", "2"); - baidu::mirana::poros::SliceConverter sliceconverter; - select_test_helper(graph_IR, &sliceconverter, {3, 4, 5, 6}); -} - -TEST(Converters, ATenSliceResnetTestConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = gen_slice_graph("1", "0", "8", "1"); - baidu::mirana::poros::SliceConverter sliceconverter; - select_test_helper(graph_IR, &sliceconverter, {1, 8, 256, 56, 56}); -} - -TEST(Converters, ATenSliceDimDynamicTestConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = gen_slice_graph("0", "2", "8", "1"); - baidu::mirana::poros::SliceConverter sliceconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({20, 16, 32}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({5, 16, 32}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({10, 16, 32}, {at::kCUDA})); - - select_test_helper(graph_IR, &sliceconverter, {10, 16, 32}, true, &prewarm_data); -} - -TEST(Converters, ATenSliceDimDynamicStartEndBothNegTestConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = gen_slice_graph("0", "-5", "-1", "1"); - baidu::mirana::poros::SliceConverter sliceconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({20, 16, 32}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({5, 16, 32}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({10, 16, 32}, {at::kCUDA})); - - select_test_helper(graph_IR, &sliceconverter, {10, 16, 32}, true, &prewarm_data); -} - -TEST(Converters, ATenSliceDimDynamicStartEndBothNoneTestConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = gen_slice_graph("0", "", "", "1"); - baidu::mirana::poros::SliceConverter sliceconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({20, 16, 32}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({5, 16, 32}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({10, 16, 32}, {at::kCUDA})); - - select_test_helper(graph_IR, &sliceconverter, {10, 16, 32}, true, &prewarm_data); -} - -TEST(Converters, ATenSliceDimDynamicTestStepConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = gen_slice_graph("0", "-8", "-1", "2"); - baidu::mirana::poros::SliceConverter sliceconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({20, 16, 32}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({5, 16, 32}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({10, 16, 32}, {at::kCUDA})); - - select_test_helper(graph_IR, &sliceconverter, {10, 16, 32}, true, &prewarm_data); -} - -TEST(Converters, ATenSliceDimNotDynamicTestConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = gen_slice_graph("1", "10", "16", "1"); - baidu::mirana::poros::SliceConverter sliceconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({20, 16, 32}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({5, 16, 32}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({10, 16, 32}, {at::kCUDA})); - - select_test_helper(graph_IR, &sliceconverter, {10, 16, 32}, true, &prewarm_data); -} - -TEST(Converters, ATenSliceDimNotDynamicStartEndBothNegTestConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = gen_slice_graph("1", "-10", "-5", "1"); - baidu::mirana::poros::SliceConverter sliceconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({20, 16, 32}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({5, 16, 32}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({10, 16, 32}, {at::kCUDA})); - - select_test_helper(graph_IR, &sliceconverter, {10, 16, 32}, true, &prewarm_data); -} - -TEST(Converters, ATenSliceDimNotDynamicStartEndBothNoneTestConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = gen_slice_graph("1", "", "", "1"); - baidu::mirana::poros::SliceConverter sliceconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({20, 16, 32}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({5, 16, 32}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({10, 16, 32}, {at::kCUDA})); - - select_test_helper(graph_IR, &sliceconverter, {10, 16, 32}, true, &prewarm_data); -} - -TEST(Converters, ATenSliceDimNotDynamicTestStepConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = gen_slice_graph("1", "-10", "-5", "2"); - baidu::mirana::poros::SliceConverter sliceconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({20, 16, 32}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({5, 16, 32}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({10, 16, 32}, {at::kCUDA})); - - select_test_helper(graph_IR, &sliceconverter, {10, 16, 32}, true, &prewarm_data); -} - -TEST(Converters, ATenSliceTStartEndBothNoneDynamicConvertsCorrectly) { - // aten::slice.t(t[] l, int? start=None, int? end=None, int step=1) -> (t[]) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int[] = aten::size(%0) - %2 : None = prim::Constant() - %3 : Device = prim::Constant[value="cuda"]() - %4 : int = prim::Constant[value=6]() - %5 : int = prim::Constant[value=1]() - %6 : int[] = aten::slice(%1, %2, %2, %5) - %7 : Tensor = aten::ones(%6, %4, %2, %3, %2) - return (%7))IR"; - baidu::mirana::poros::SliceConverter sliceconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({5, 6, 7, 8}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({4, 5, 6, 7}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({4, 5, 6, 7}, {at::kCUDA})); - - select_test_helper(graph_IR, &sliceconverter, {4, 5, 6, 7}, true, &prewarm_data); -} - -TEST(Converters, ATenSliceTStartEndDynamicConvertsCorrectly) { - // aten::slice.t(t[] l, int? start=None, int? end=None, int step=1) -> (t[]) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %start : int = prim::Constant[value=1]() - %end : int = prim::Constant[value=3]() - %1 : int = prim::Constant[value=1]() - %2 : None = prim::Constant() - %3 : int[] = aten::size(%0) - %4 : int[] = aten::slice(%3, %start, %end, %1) - %5 : Device = prim::Constant[value="cuda"]() - %6 : int = prim::Constant[value=6]() - %7 : Tensor = aten::ones(%4, %6, %2, %5, %2) - return (%7))IR"; - baidu::mirana::poros::SliceConverter sliceconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({5, 6, 7, 8}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({4, 5, 6, 7}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({4, 5, 6, 7}, {at::kCUDA})); - - select_test_helper(graph_IR, &sliceconverter, {4, 5, 6, 7}, true, &prewarm_data); -} - -TEST(Converters, ATenSliceTStartEndNegDynamicConvertsCorrectly) { - // aten::slice.t(t[] l, int? start=None, int? end=None, int step=1) -> (t[]) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %start : int = prim::Constant[value=-3]() - %end : int = prim::Constant[value=-1]() - %1 : int = prim::Constant[value=1]() - %2 : None = prim::Constant() - %3 : int[] = aten::size(%0) - %4 : int[] = aten::slice(%3, %start, %end, %1) - %5 : Device = prim::Constant[value="cuda"]() - %6 : int = prim::Constant[value=6]() - %7 : Tensor = aten::ones(%4, %6, %2, %5, %2) - return (%7))IR"; - baidu::mirana::poros::SliceConverter sliceconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({5, 6, 7, 8}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({4, 5, 6, 7}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({4, 5, 6, 7}, {at::kCUDA})); - - select_test_helper(graph_IR, &sliceconverter, {4, 5, 6, 7}, true, &prewarm_data); -} - -TEST(Converters, ATenSliceTStartEndStepDynamicConvertsCorrectly) { - // aten::slice.t(t[] l, int? start=None, int? end=None, int step=1) -> (t[]) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %start : int = prim::Constant[value=0]() - %end : int = prim::Constant[value=3]() - %step : int = prim::Constant[value=2]() - %1 : int = prim::Constant[value=1]() - %2 : None = prim::Constant() - %3 : int[] = aten::size(%0) - %4 : int[] = aten::slice(%3, %start, %end, %step) - %5 : Device = prim::Constant[value="cuda"]() - %6 : int = prim::Constant[value=6]() - %7 : Tensor = aten::ones(%4, %6, %2, %5, %2) - return (%7))IR"; - baidu::mirana::poros::SliceConverter sliceconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({5, 6, 7, 8}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({4, 5, 6, 7}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({4, 5, 6, 7}, {at::kCUDA})); - - select_test_helper(graph_IR, &sliceconverter, {4, 5, 6, 7}, true, &prewarm_data); -} - -TEST(Converters, ATenSliceFromSizeStartDynamicConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=1]() - %2 : int = aten::size(%0, %1) - %3 : int = prim::Constant[value=3]() - %4 : int = aten::floordiv(%2, %3) - %end : None = prim::Constant() - %step : int = prim::Constant[value=1]() - %5 : Tensor = aten::slice(%0, %1, %4, %end, %step) - return (%5))IR"; - baidu::mirana::poros::SliceConverter sliceconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({5, 10, 7, 8}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({4, 5, 6, 7}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({4, 5, 6, 7}, {at::kCUDA})); - - select_test_helper(graph_IR, &sliceconverter, {4, 5, 6, 7}, true, &prewarm_data); -} - -TEST(Converters, ATenSliceFromSizeEndDynamicConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=1]() - %2 : int = aten::size(%0, %1) - %3 : int = prim::Constant[value=3]() - %4 : int = aten::floordiv(%2, %3) - %start : None = prim::Constant() - %step : int = prim::Constant[value=1]() - %5 : Tensor = aten::slice(%0, %1, %start, %4, %step) - return (%5))IR"; - baidu::mirana::poros::SliceConverter sliceconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({5, 10, 7, 8}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({4, 5, 6, 7}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({4, 5, 6, 7}, {at::kCUDA})); - - select_test_helper(graph_IR, &sliceconverter, {4, 5, 6, 7}, true, &prewarm_data); -} - -TEST(Converters, ATenSliceFromSizeStartEndDynamicConvertsCorrectly) { - // aten::slice.Tensor(Tensor(a) self, int dim=0, int? start=None, int? end=None, int step=1) -> Tensor(a) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=1]() - %2 : int = aten::size(%0, %1) - %3 : int = prim::Constant[value=2]() - %4 : int = prim::Constant[value=5]() - %5 : int = aten::floordiv(%2, %3) - %6 : int = aten::floordiv(%2, %4) - %step : int = prim::Constant[value=1]() - %5 : Tensor = aten::slice(%0, %1, %6, %5, %step) - return (%5))IR"; - baidu::mirana::poros::SliceConverter sliceconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({5, 10, 7, 8}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({4, 5, 6, 7}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({4, 5, 6, 7}, {at::kCUDA})); - - select_test_helper(graph_IR, &sliceconverter, {4, 5, 6, 7}, true, &prewarm_data); -} - -TEST(Converters, ATenNarrowScalarConvertsCorrectly) { - // aten::narrow(Tensor(a) self, int dim, int start, int length) -> Tensor(a) - const auto graph_IR = gen_narrow_graph("2", "0", "2", true); - baidu::mirana::poros::NarrowConverter narrowconverter; - select_test_helper(graph_IR, &narrowconverter, {4, 4, 4, 4}); -} - -TEST(Converters, ATenNarrowScalarNegtiveStartConvertsCorrectly) { - // aten::narrow(Tensor(a) self, int dim, int start, int length) -> Tensor(a) - const auto graph_IR = gen_narrow_graph("2", "-3", "2", true); - baidu::mirana::poros::NarrowConverter narrowconverter; - select_test_helper(graph_IR, &narrowconverter, {4, 4, 4, 4}); -} - -TEST(Converters, ATenNarrowScalarNegtiveDimConvertsCorrectly) { - // aten::narrow(Tensor(a) self, int dim, int start, int length) -> Tensor(a) - const auto graph_IR = gen_narrow_graph("-2", "0", "2", true); - baidu::mirana::poros::NarrowConverter narrowconverter; - select_test_helper(graph_IR, &narrowconverter, {4, 4, 4, 4}); -} - -TEST(Converters, ATenNarrowScalarNegtiveDimStartConvertsCorrectly) { - // aten::narrow(Tensor(a) self, int dim, int start, int length) -> Tensor(a) - const auto graph_IR = gen_narrow_graph("-2", "-3", "2", true); - baidu::mirana::poros::NarrowConverter narrowconverter; - select_test_helper(graph_IR, &narrowconverter, {4, 4, 4, 4}); -} - -TEST(Converters, ATenSplitFixedTensorsConvertsCorrectly) { - // aten::split.Tensor(Tensor(a) self, int split_size, int dim=0) -> Tensor(a)[] - const auto graph_IR = R"IR( - graph(%1 : Tensor): - %2 : int = prim::Constant[value=3]() - %3 : int = prim::Constant[value=0]() - %4 : Tensor[] = aten::split(%1, %2, %3) - %5 : Tensor, %6 : Tensor = prim::ListUnpack(%4) - return (%5, %6))IR"; - baidu::mirana::poros::SplitConverter splitconverter; - split_test_helper(graph_IR, &splitconverter, {6, 4, 3, 1}, 2); -} - -TEST(Converters, ATenSplitUnfixedTensorsConvertsCorrectly) { - // aten::split.Tensor(Tensor(a) self, int split_size, int dim=0) -> Tensor(a)[] - const auto graph_IR = R"IR( - graph(%1 : Tensor): - %2 : int = prim::Constant[value=2]() - %3 : int = prim::Constant[value=1]() - %4 : Tensor[] = aten::split(%1, %2, %3) - %5 : Tensor, %6 : Tensor, %7 : Tensor = prim::ListUnpack(%4) - return (%5, %6, %7))IR"; - baidu::mirana::poros::SplitConverter splitconverter; - split_test_helper(graph_IR, &splitconverter, {4, 5, 3, 1}, 3); -} - -TEST(Converters, ATenSplitWithSizeDoubleTensorsConvertsCorrectly) { - // aten::split_with_sizes(Tensor(a) self, int[] split_sizes, int dim=0) -> Tensor(a)[] - const auto graph_IR = R"IR( - graph(%1 : Tensor): - %2 : int[] = prim::Constant[value=[4, 2]]() - %3 : int = prim::Constant[value=0]() - %4 : Tensor[] = aten::split_with_sizes(%1, %2, %3) - %5 : Tensor, %6 : Tensor = prim::ListUnpack(%4) - return (%5, %6))IR"; - baidu::mirana::poros::SplitConverter splitconverter; - split_test_helper(graph_IR, &splitconverter, {6, 4, 3, 1}, 2); -} - -TEST(Converters, ATenSplitWithSizeTrippleTensorsConvertsCorrectly) { - // aten::split_with_sizes(Tensor(a) self, int[] split_sizes, int dim=0) -> Tensor(a)[] - const auto graph_IR = R"IR( - graph(%1 : Tensor): - %2 : int[] = prim::Constant[value=[5, 1, 2]]() - %3 : int = prim::Constant[value=1]() - %4 : Tensor[] = aten::split_with_sizes(%1, %2, %3) - %5 : Tensor, %6 : Tensor, %7 : Tensor = prim::ListUnpack(%4) - return (%5, %6, %7))IR"; - baidu::mirana::poros::SplitConverter splitconverter; - split_test_helper(graph_IR, &splitconverter, {2, 8, 3, 1}, 3); -} - -TEST(Converters, ATenUnbindTensorsConvertsCorrectly) { - // aten::unbind.int(Tensor(a) self, int dim=0) -> Tensor(a)[] - const auto graph_IR = R"IR( - graph(%1 : Tensor): - %2 : int = prim::Constant[value=1]() - %3 : Tensor[] = aten::unbind(%1, %2) - %4 : Tensor, %5 : Tensor, %6 : Tensor = prim::ListUnpack(%3) - return (%4, %5, %6))IR"; - baidu::mirana::poros::SplitConverter splitconverter; - split_test_helper(graph_IR, &splitconverter, {2, 3, 4}, 3); -} - -TEST(Converters, EmbeddingConverterCorrectly) { - const auto graph_IR = R"IR( - graph(%weight.1 : Tensor, - %input.1 : Tensor): - %7 : bool = prim::Constant[value=0]() - %6 : int = prim::Constant[value=-1]() - %9 : Tensor = aten::embedding(%weight.1, %input.1, %6, %7, %7) - return (%9))IR"; - - baidu::mirana::poros::EmbeddingConverter embeddingconverter; - embedding_test_helper(graph_IR, &embeddingconverter); -} - -static void gather_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - const std::vector& input_data, - bool is_dynamic = false, - std::vector>* prewarm_data = nullptr) { - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = is_dynamic; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output, prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenGatherConverterCorrectly) { - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : bool = prim::Constant[value=0]() - %3 : int = prim::Constant[value=1]() - %4 : Tensor = aten::gather(%0, %3, %1, %2) - return (%4))IR"; - - std::vector input_data; - auto input = at::randn({3, 4, 5, 6}, {at::kCUDA}); - auto options_pyt = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kInt64); - auto index = at::randint(0, 2, {3, 4, 5, 6}, options_pyt); - - input_data.push_back(input); - input_data.push_back(index); - - baidu::mirana::poros::GatherConverter gatherconverter; - gather_test_helper(graph_IR, &gatherconverter, input_data, false); -} - -TEST(Converters, ATenGatherNegtiveDimConverterCorrectly) { - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : bool = prim::Constant[value=0]() - %3 : int = prim::Constant[value=-1]() - %4 : Tensor = aten::gather(%0, %3, %1, %2) - return (%4))IR"; - - std::vector input_data; - auto input = at::randn({3, 4, 5, 6}, {at::kCUDA}); - auto options_pyt = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kInt64); - auto index = at::randint(0, 2, {3, 4, 5, 6}, options_pyt); - - input_data.push_back(input); - input_data.push_back(index); - - baidu::mirana::poros::GatherConverter gatherconverter; - gather_test_helper(graph_IR, &gatherconverter, input_data, false); -} - -TEST(Converters, ATenGatherDynamicConverterCorrectly) { - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : bool = prim::Constant[value=0]() - %3 : int = prim::Constant[value=-1]() - %4 : Tensor = aten::gather(%0, %3, %1, %2) - return (%4))IR"; - - std::vector> prewarm_data = {{}, {}, {}}; - - auto options_pyt = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kInt64); - // max - prewarm_data[0].push_back(at::randn({4, 5, 6, 7}, {at::kCUDA})); - prewarm_data[0].push_back(at::randint(0, 3, {4, 5, 6, 7}, options_pyt)); - // min - prewarm_data[1].push_back(at::randn({3, 4, 5, 6}, {at::kCUDA})); - prewarm_data[1].push_back(at::randint(0, 2, {3, 4, 5, 6}, options_pyt)); - // opt - prewarm_data[2].push_back(at::randn({3, 4, 5, 6}, {at::kCUDA})); - prewarm_data[2].push_back(at::randint(0, 2, {3, 4, 5, 6}, options_pyt)); - - std::vector input_data; - auto input = at::randn({3, 4, 5, 6}, {at::kCUDA}); - auto index = at::randint(0, 2, {3, 4, 5, 6}, options_pyt); - - input_data.push_back(input); - input_data.push_back(index); - - baidu::mirana::poros::GatherConverter gatherconverter; - gather_test_helper(graph_IR, &gatherconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenMaskedfillScalarValueConverterCorrectly) { -// aten::masked_fill.Scalar(Tensor self, Tensor mask, Scalar value) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : int = prim::Constant[value=-1]() - %3 : Tensor = aten::masked_fill(%0, %1, %2) - return (%3))IR"; - - std::vector input_data; - input_data.push_back(at::randn({2, 2}, {at::kCUDA})); - auto options_pyt = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kBool); - input_data.push_back(torch::tensor({false, true, false, true}, options_pyt).reshape({2, 2})); - - baidu::mirana::poros::MaskedFillConverter maskedfillconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = false; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &maskedfillconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenMaskedfillScalarValueDynamicConverterCorrectly) { -// aten::masked_fill.Scalar(Tensor self, Tensor mask, Scalar value) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : int = prim::Constant[value=-1]() - %3 : Tensor = aten::masked_fill(%0, %1, %2) - return (%3))IR"; - - std::vector> prewarm_data = {{}, {}, {}}; - - auto options_pyt = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kBool); - // max - prewarm_data[0].push_back(at::randn({4, 2}, {at::kCUDA})); - prewarm_data[0].push_back(torch::tensor({false, true, false, true, false, true, false, true}, options_pyt).reshape({4, 2})); - // min - prewarm_data[1].push_back(at::randn({1, 2}, {at::kCUDA})); - prewarm_data[1].push_back(torch::tensor({false, true}, options_pyt).reshape({1, 2})); - // opt - prewarm_data[2].push_back(at::randn({2, 2}, {at::kCUDA})); - prewarm_data[2].push_back(torch::tensor({false, true, false, true}, options_pyt).reshape({2, 2})); - - std::vector input_data; - input_data.push_back(at::randn({2, 2}, {at::kCUDA})); - input_data.push_back(torch::tensor({false, true, false, true}, options_pyt).reshape({2, 2})); - - baidu::mirana::poros::MaskedFillConverter maskedfillconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = true; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &maskedfillconverter, - input_data, graph_output, poros_output, &prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenMaskedfillScalarValueDynamicMoreConverterCorrectly) { -// aten::masked_fill.Scalar(Tensor self, Tensor mask, Scalar value) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : int = prim::Constant[value=-1]() - %3 : Tensor = aten::masked_fill(%0, %1, %2) - return (%3))IR"; - - std::vector> prewarm_data = {{}, {}, {}}; - - auto options_pyt = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kBool); - // max - prewarm_data[0].push_back(at::randn({4, 2}, {at::kCUDA})); - prewarm_data[0].push_back(torch::tensor({false, true}, options_pyt).reshape({2})); - // min - prewarm_data[1].push_back(at::randn({1, 2}, {at::kCUDA})); - prewarm_data[1].push_back(torch::tensor({false, true}, options_pyt).reshape({2})); - // opt - prewarm_data[2].push_back(at::randn({2, 2}, {at::kCUDA})); - prewarm_data[2].push_back(torch::tensor({true, true}, options_pyt).reshape({2})); - - std::vector input_data; - input_data.push_back(at::randn({2, 2}, {at::kCUDA})); - input_data.push_back(torch::tensor({false, true}, options_pyt).reshape({2})); - - baidu::mirana::poros::MaskedFillConverter maskedfillconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = true; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &maskedfillconverter, - input_data, graph_output, poros_output, &prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenMaskedfillTensorValueConverterCorrectly) { -// aten::masked_fill.Tensor(Tensor self, Tensor mask, Tensor value) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %false : bool = prim::Constant[value=0]() - %2 : int = prim::Constant[value=2]() - %device : Device = prim::Constant[value="cuda:0"]() - %type : int = prim::dtype(%0) - %value : Tensor = aten::tensor(%2, %type, %device, %false) - %4 : Tensor = aten::masked_fill(%0, %1, %value) - return (%4))IR"; - - std::vector input_data; - input_data.push_back(at::randn({2, 2}, {at::kCUDA})); - auto options_pyt = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kBool); - input_data.push_back(torch::tensor({false, true, false, true}, options_pyt).reshape({2, 2})); - - baidu::mirana::poros::MaskedFillConverter maskedfillconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = false; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &maskedfillconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - - -TEST(Converters, ATenIndexOneDimConverterCorrectly) { -// aten::index.Tensor(Tensor self, Tensor?[] indices) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : Tensor?[] = prim::ListConstruct(%0) - %3 : Tensor = aten::index(%1, %2) - return (%3))IR"; - - std::vector input_data; - auto options_pyt = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kLong); - input_data.push_back(at::randint(0, 3, {2, 2}, options_pyt)); // indices - input_data.push_back(at::randint(0, 10, {3, 4, 5}, options_pyt)); // self - - baidu::mirana::poros::IndexConverter indexconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = false; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &indexconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenIndexOneDimDynamicConverterCorrectly) { -// aten::index.Tensor(Tensor self, Tensor?[] indices) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : Tensor?[] = prim::ListConstruct(%0) - %3 : Tensor = aten::index(%1, %2) - return (%3))IR"; - - std::vector input_data; - auto options_pyt = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kLong); - input_data.push_back(at::randint(0, 3, {2, 2}, options_pyt)); // indices - input_data.push_back(at::randint(0, 10, {3, 4, 5}, options_pyt)); // self - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randint(0, 4, {3, 4}, options_pyt)); // indices - prewarm_data[0].push_back(at::randint(0, 10, {4, 5, 6}, options_pyt)); // self - prewarm_data[1].push_back(at::randint(0, 3, {2, 2}, options_pyt)); // indices - prewarm_data[1].push_back(at::randint(0, 10, {3, 4, 5}, options_pyt)); // self - prewarm_data[2].push_back(at::randint(0, 3, {2, 2}, options_pyt)); // indices - prewarm_data[2].push_back(at::randint(0, 10, {3, 4, 5}, options_pyt)); // self - - baidu::mirana::poros::IndexConverter indexconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = true; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &indexconverter, - input_data, graph_output, poros_output, &prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenIndexPutConverterCorrectly) { -//aten::index_put(Tensor self, Tensor?[] indices, Tensor values, bool accumulate=False) -> Tensor" - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor, %2 : Tensor, %3 : Tensor): - %false : bool = prim::Constant[value=0]() - %none : NoneType = prim::Constant() - %zeros : Tensor = aten::zeros_like(%0, %none, %none, %none, %none, %none) - %index : Tensor?[] = prim::ListConstruct(%1, %2) - %out : Tensor = aten::index_put(%zeros, %index, %3, %false) - return (%out))IR"; - - std::vector input_data; - auto options_pyt_long = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kLong); - auto options_pyt_float = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kFloat); - input_data.push_back(at::zeros({2, 5}, options_pyt_float)); - input_data.push_back(torch::tensor({0, 0, 1, 1}, options_pyt_long)); - input_data.push_back(torch::tensor({0, 2, 1, 3}, options_pyt_long)); - input_data.push_back(torch::tensor({1, 2, 3, 4}, options_pyt_float)); - - baidu::mirana::poros::IndexPutConverter indexputconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = false; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &indexputconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -//2022.10.19 踩坑记录:不要对indexput 这个singular的IR 进行非dynamic的单测, -//因为这个graph在static的情况下,会在数据预热阶段, 直接全图计算出结果,生成一个constant结果给tensorrt, -//直接导致单测无法通过。 -TEST(Converters, ATenIndexPutConverterSingularValueDynamicCorrectly) { -//aten::index_put(Tensor self, Tensor?[] indices, Tensor values, bool accumulate=False) -> Tensor" - const auto graph_IR = gen_indexput_with_singular_value_graph(); - std::vector input_data; - auto options_pyt_float = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kFloat); - input_data.push_back(at::zeros({1, 16, 64}, options_pyt_float)); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::zeros({1, 30, 64}, options_pyt_float)); - prewarm_data[1].push_back(at::zeros({1, 8, 64}, options_pyt_float)); - prewarm_data[2].push_back(at::zeros({1, 20, 64}, options_pyt_float)); - - baidu::mirana::poros::IndexPutConverter indexputconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = true; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &indexputconverter, - input_data, graph_output, poros_output, &prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenIndexPutConverterDynamicCorrectly) { -//aten::index_put(Tensor self, Tensor?[] indices, Tensor values, bool accumulate=False) -> Tensor" - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor, %2 : Tensor, %3 : Tensor): - %false : bool = prim::Constant[value=0]() - %none : NoneType = prim::Constant() - %zeros : Tensor = aten::zeros_like(%0, %none, %none, %none, %none, %none) - %index : Tensor?[] = prim::ListConstruct(%1, %2) - %out : Tensor = aten::index_put(%zeros, %index, %3, %false) - return (%out))IR"; - - std::vector input_data; - auto options_pyt_long = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kLong); - auto options_pyt_float = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kFloat); - input_data.push_back(at::zeros({2, 5}, options_pyt_float)); - input_data.push_back(torch::tensor({0, 0, 1, 1}, options_pyt_long)); - input_data.push_back(torch::tensor({0, 2, 1, 3}, options_pyt_long)); - input_data.push_back(torch::tensor({1, 2, 3, 4}, options_pyt_float)); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::zeros({8, 5}, options_pyt_float)); // indices - prewarm_data[0].push_back(torch::tensor({2, 3, 3, 4}, options_pyt_long)); - prewarm_data[0].push_back(torch::tensor({0, 2, 1, 3}, options_pyt_long)); - prewarm_data[0].push_back(torch::tensor({8, 8, 8, 8}, options_pyt_float)); - - prewarm_data[1].push_back(at::zeros({2, 5}, options_pyt_float)); // indices - prewarm_data[1].push_back(torch::tensor({0, 0, 1, 1}, options_pyt_long)); - prewarm_data[1].push_back(torch::tensor({0, 2, 1, 3}, options_pyt_long)); - prewarm_data[1].push_back(torch::tensor({3, 3, 3, 3}, options_pyt_float)); - - prewarm_data[2].push_back(at::zeros({4, 5}, options_pyt_float)); // indices - prewarm_data[2].push_back(torch::tensor({0, 0, 1, 1}, options_pyt_long)); - prewarm_data[2].push_back(torch::tensor({0, 2, 1, 3}, options_pyt_long)); - prewarm_data[2].push_back(torch::tensor({1, 2, 3, 4}, options_pyt_float)); - - baidu::mirana::poros::IndexPutConverter indexputconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = true; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &indexputconverter, - input_data, graph_output, poros_output, &prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenIndexPutConverterDynamicFromCopyCorrectly) { -//aten::index_put(Tensor self, Tensor?[] indices, Tensor values, bool accumulate=False) -> Tensor" - const auto graph_IR = gen_indexput_graph("21"); - - std::vector input_data; - auto options_pyt_float = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kFloat); - input_data.push_back(at::ones({8, 16, 64, 16, 16}, options_pyt_float)); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::ones({64, 16, 64, 16, 16}, options_pyt_float)); - prewarm_data[1].push_back(at::ones({8, 16, 64, 16, 16}, options_pyt_float)); - prewarm_data[2].push_back(at::ones({32, 16, 64, 16, 16}, options_pyt_float)); - - baidu::mirana::poros::IndexPutConverter indexputconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = true; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &indexputconverter, - input_data, graph_output, poros_output, &prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenIndexPutConverterStaticFromCopyCorrectly) { -//aten::index_put(Tensor self, Tensor?[] indices, Tensor values, bool accumulate=False) -> Tensor" - const auto graph_IR = gen_indexput_graph("21"); - - std::vector input_data; - auto options_pyt_float = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kFloat); - input_data.push_back(at::ones({1, 16, 64, 56, 56}, options_pyt_float)); - - baidu::mirana::poros::IndexPutConverter indexputconverter; - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = false; - - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &indexputconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenScatterConverterCorrectly) { -// aten::scatter.value(Tensor self, int dim, Tensor index, Scalar value) -> (Tensor) - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : int = prim::Constant[value=1]() - %3 : float = prim::Constant[value=2.5]() - %4 : Tensor = aten::scatter(%0, %2, %1, %3) - return (%4))IR"; - - std::vector input_data; - auto options_pyt_long = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kLong); - auto options_pyt_float = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kFloat); - input_data.push_back(at::zeros({2, 4}, options_pyt_float)); - input_data.push_back(torch::tensor({{0, 1, 2, 0}, {1, 2, 0, 3}}, options_pyt_long)); - - baidu::mirana::poros::ScatterConverter scatterconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = false; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &scatterconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenScatterSelfValueDiffTypeConverterCorrectly) { -// aten::scatter.value(Tensor self, int dim, Tensor index, Scalar value) -> (Tensor) - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : int = prim::Constant[value=1]() - %3 : float = prim::Constant[value=2.5]() - %4 : Tensor = aten::scatter(%0, %2, %1, %3) - return (%4))IR"; - - std::vector input_data; - auto options_pyt_long = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kLong); - auto options_pyt_int = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kInt); - input_data.push_back(at::zeros({2, 4}, options_pyt_int)); - input_data.push_back(torch::tensor({{0, 1, 2, 3}, {1, 2, 0, 3}}, options_pyt_long)); - - baidu::mirana::poros::ScatterConverter scatterconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = false; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &scatterconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenScatterSelfIndexDiffShapeConverterCorrectly) { -// aten::scatter.value(Tensor self, int dim, Tensor index, Scalar value) -> (Tensor) -// Index tensor 和 self tensor 的shape可以不一致但是rank(dim数)必须一致 - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : int = prim::Constant[value=0]() - %3 : float = prim::Constant[value=2.5]() - %4 : Tensor = aten::scatter(%0, %2, %1, %3) - return (%4))IR"; - - std::vector input_data; - auto options_pyt_long = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kLong); - auto options_pyt_float = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kFloat); - input_data.push_back(at::zeros({2, 4}, options_pyt_float)); - input_data.push_back(torch::tensor({{0}}, options_pyt_long)); - - baidu::mirana::poros::ScatterConverter scatterconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = false; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &scatterconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenScatterDynamicConverterCorrectly) { -// aten::scatter.value(Tensor self, int dim, Tensor index, Scalar value) -> (Tensor) - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : int = prim::Constant[value=1]() - %3 : float = prim::Constant[value=2.5]() - %4 : Tensor = aten::scatter(%0, %2, %1, %3) - return (%4))IR"; - - std::vector input_data; - auto options_pyt_long = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kLong); - auto options_pyt_float = torch::TensorOptions().device(torch::kCUDA, 0).dtype(torch::kFloat); - input_data.push_back(at::zeros({3, 4}, options_pyt_float)); - input_data.push_back(torch::tensor({{0, 1}, {1, 2}}, options_pyt_long)); - - std::vector> prewarm_data = {{}, {}, {}}; - // max - prewarm_data[0].push_back(at::randint(0, 4, {3, 4}, options_pyt_float)); - prewarm_data[0].push_back(at::randint(0, 2, {3, 4}, options_pyt_long)); - // min - prewarm_data[1].push_back(at::randint(0, 4, {2, 3}, options_pyt_float)); - prewarm_data[1].push_back(at::randint(0, 2, {1, 1}, options_pyt_long)); - // opt - prewarm_data[2].push_back(at::randint(0, 4, {3, 4}, options_pyt_float)); - prewarm_data[2].push_back(at::randint(0, 2, {1, 1}, options_pyt_long)); - - baidu::mirana::poros::ScatterConverter scatterconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = true; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &scatterconverter, - input_data, graph_output, poros_output, &prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -static void chunk_test_helper(const std::string& graph_IR, - std::vector shape, - const int& output_num) { - std::vector input_data; - input_data.push_back(at::randn(shape, {at::kCUDA})); - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - baidu::mirana::poros::ChunkConverter chunkconverter; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &chunkconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(output_num, graph_output.size()); - ASSERT_EQ(graph_output.size(), poros_output.size()); - for (size_t i = 0; i < graph_output.size(); i++) { - ASSERT_TRUE(graph_output[i].equal(poros_output[i])); - } -} - -TEST(Converters, PrimConstantChunkTwoOutputsConverterCorrectly) { - // prim::chunk - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : Tensor, %2 : Tensor = prim::ConstantChunk[chunks=2, dim=-1](%0) - return (%1, %2))IR"; - chunk_test_helper(graph_IR, {5, 6, 8}, 2); -} - -TEST(Converters, PrimConstantChunkThreeOutputsConverterCorrectly) { - // prim::chunk - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : Tensor, %2 : Tensor, %3 : Tensor = prim::ConstantChunk[chunks=3, dim=0](%0) - return (%1, %2, %3))IR"; - chunk_test_helper(graph_IR, {11, 6, 8}, 3); -} - -TEST(Converters, PrimConstantChunkFourOutputsConverterCorrectly) { - // prim::chunk - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : Tensor, %2 : Tensor, %3 : Tensor, %4 : Tensor = prim::ConstantChunk[chunks=4, dim=1](%0) - return (%1, %2, %3, %4))IR"; - chunk_test_helper(graph_IR, {7, 13, 8}, 4); -} \ No newline at end of file diff --git a/poros/unittest/converter/shape_handle_test.cpp b/poros/unittest/converter/shape_handle_test.cpp deleted file mode 100644 index b2685046af2..00000000000 --- a/poros/unittest/converter/shape_handle_test.cpp +++ /dev/null @@ -1,119 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file shape_handle_test.cpp -* @author tianshaoqing@baidu.com -* @date Tues Jul 27 14:24:21 CST 2022 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/shape_handle.h" -#include "poros/util/test_util.h" - -static void shape_handle_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - std::vector shape, - bool is_dynamic = false, - std::vector>* prewarm_data = nullptr) { - std::vector input_data; - input_data.push_back(at::randn(shape, {at::kCUDA})); - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = is_dynamic; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output, prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenShapeAsTensorConvertsCorrectly) { - // aten::_shape_as_tensor(Tensor self) -> (Tensor) - // aten::_shape_as_tensor output tensor is default on cpu. - // To keep all data on same device, need to add aten::to.device. - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : Tensor = aten::_shape_as_tensor(%0) - %2 : Device = prim::Constant[value="cuda"]() - %3 : int = prim::Constant[value=3]() - %4 : bool = prim::Constant[value=0]() - %5 : None = prim::Constant() - %6 : Tensor = aten::to(%1, %2, %3, %4, %4, %5) - return (%6))IR"; - baidu::mirana::poros::ShapeastensorConverter shapeastensorconverter; - shape_handle_test_helper(graph_IR, &shapeastensorconverter, {4, 5, 3, 1}); -} - -TEST(Converters, ATenShapeAsTensorDynamicConvertsCorrectly) { - // aten::_shape_as_tensor(Tensor self) -> (Tensor) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : Tensor = aten::_shape_as_tensor(%0) - %2 : Device = prim::Constant[value="cuda"]() - %3 : int = prim::Constant[value=3]() - %4 : bool = prim::Constant[value=0]() - %5 : None = prim::Constant() - %6 : Tensor = aten::to(%1, %2, %3, %4, %4, %5) - return (%6))IR"; - baidu::mirana::poros::ShapeastensorConverter shapeastensorconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({5, 10, 7, 8}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({4, 5, 6, 7}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({4, 5, 6, 7}, {at::kCUDA})); - - shape_handle_test_helper(graph_IR, &shapeastensorconverter, {4, 5, 6, 7}, true, &prewarm_data); -} - -// aten::len.Tensor(Tensor t) -> (int) -// aten::len.t(t[] a) -> (int) -TEST(Converters, ATenLenDynamicConvertsCorrectly) { - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = aten::len(%0) - %2 : NoneType = prim::Constant() - %3 : bool = prim::Constant[value=0]() - %4 : Device = prim::Constant[value="cuda:0"]() - %5 : Tensor = aten::tensor(%1, %2, %4, %3) - return (%5))IR"; - - baidu::mirana::poros::LenConverter lenconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({7, 2}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({3, 2}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({5, 2}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::ones({7, 2}, {at::kCUDA})); - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = true; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &lenconverter, - input_data, graph_output, poros_output, &prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} \ No newline at end of file diff --git a/poros/unittest/converter/shuffle_test.cpp b/poros/unittest/converter/shuffle_test.cpp deleted file mode 100644 index 8c49c456d43..00000000000 --- a/poros/unittest/converter/shuffle_test.cpp +++ /dev/null @@ -1,313 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file shuffle_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/util/test_util.h" -#include "poros/converter/gpu/shuffle.h" - -static void shuffle_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - std::vector shape){ - std::vector input_data; - input_data.push_back(at::randn(shape, {at::kCUDA})); - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -static void shuffle_dy_test_helper(const std::string& graph_IR, - const std::vector& input_data, - baidu::mirana::poros::IConverter* converter, - bool is_dynamic = false, - std::vector>* prewarm_data = nullptr) { - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = is_dynamic; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output, prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -std::string gen_double_int_graph(const std::string& op, - const std::string& first_int, - const std::string& second_int) { - return R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=)IR" + first_int + R"IR(]() - %2 : int = prim::Constant[value=)IR" + second_int + R"IR(]() - %3 : Tensor = aten::)IR" + op + R"IR((%0, %1, %2) - return (%3))IR"; -} - -std::string gen_int_list_graph(const std::string& op, const std::string& int_list) { - return R"IR( - graph(%0 : Tensor): - %1 : int[] = prim::Constant[value=[)IR" + int_list + R"IR(]]() - %2 : Tensor = aten::)IR" + op + R"IR((%0, %1) - return (%2))IR"; -} - -std::string gen_pixel_shuffle_graph(const std::string& upscale_factor) { - return R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=)IR" + upscale_factor + R"IR(]() - %2 : Tensor = aten::pixel_shuffle(%0, %1) - return (%2))IR"; -} - -TEST(Converters, ATenTransposeConvertsCorrectly) { - // aten::transpose.int(Tensor(a) self, int dim0, int dim1) -> Tensor(a) - const auto graph_IR = gen_double_int_graph("transpose", "1", "2"); - baidu::mirana::poros::TransposeConverter transposeconverter; - shuffle_test_helper(graph_IR, &transposeconverter, {2, 3, 4}); -} - -TEST(Converters, ATenTransposeNegaiveConvertsCorrectly) { - // aten::transpose.int(Tensor(a) self, int dim0, int dim1) -> Tensor(a) - const auto graph_IR = gen_double_int_graph("transpose", "-1", "-3"); - baidu::mirana::poros::TransposeConverter transposeconverter; - shuffle_test_helper(graph_IR, &transposeconverter, {2, 3, 4, 5, 6}); -} - -TEST(Converters, ATenViewConvertsCorrectly) { - // aten::view(Tensor(a) self, int[] size) -> Tensor(a) - const auto graph_IR = gen_int_list_graph("view", "1, 6"); - baidu::mirana::poros::PermuteViewConverter permuteviewconverter; - shuffle_test_helper(graph_IR, &permuteviewconverter, {2, 3}); -} - -TEST(Converters, ATenViewNegtiveConvertsCorrectly) { - // aten::view(Tensor(a) self, int[] size) -> Tensor(a) - const auto graph_IR = gen_int_list_graph("view", "-1, 8"); - baidu::mirana::poros::PermuteViewConverter permuteviewconverter; - shuffle_test_helper(graph_IR, &permuteviewconverter, {4, 4}); -} - -TEST(Converters, ATenPermuteConvertsCorrectly) { - // aten::permute(Tensor(a) self, int[] dims) -> Tensor(a) - const auto graph_IR = gen_int_list_graph("permute", "1, 0"); - baidu::mirana::poros::PermuteViewConverter permuteviewconverter; - shuffle_test_helper(graph_IR, &permuteviewconverter, {2, 3}); -} - -TEST(Converters, ATenPermute3DConvertsCorrectly) { - // aten::permute(Tensor(a) self, int[] dims) -> Tensor(a) - const auto graph_IR = gen_int_list_graph("permute", "1, 2, 0"); - baidu::mirana::poros::PermuteViewConverter permuteviewconverter; - shuffle_test_helper(graph_IR, &permuteviewconverter, {1, 2, 3}); -} - -TEST(Converters, ATenPermute5DConvertsCorrectly) { - // aten::permute(Tensor(a) self, int[] dims) -> Tensor(a) - const auto graph_IR = gen_int_list_graph("permute", "3, 1, 0, 2, 4"); - baidu::mirana::poros::PermuteViewConverter permuteviewconverter; - shuffle_test_helper(graph_IR, &permuteviewconverter, {2, 3, 4, 5, 1}); -} - -TEST(Converters, ATenReshapeConvertsCorrectly) { - // aten::reshape(Tensor(a) self, int[] shape) -> Tensor(a) - const auto graph_IR = gen_int_list_graph("reshape", "3, 2"); - baidu::mirana::poros::ReshapeConverter reshapeconverter; - shuffle_test_helper(graph_IR, &reshapeconverter, {2, 3}); -} - -TEST(Converters, ATenReshapeNegtiveConvertsCorrectly) { - // aten::reshape(Tensor(a) self, int[] shape) -> Tensor(a) - const auto graph_IR = gen_int_list_graph("reshape", "-1, 8"); - baidu::mirana::poros::ReshapeConverter reshapeconverter; - shuffle_test_helper(graph_IR, &reshapeconverter, {4, 4}); -} - -TEST(Converters, ATenFlattenConvertsCorrectly) { - // aten::flatten.using_ints(Tensor(a) self, int start_dim=0, int end_dim=-1) -> Tensor(a) - const auto graph_IR = gen_double_int_graph("flatten", "0", "-1"); - baidu::mirana::poros::FlattenConverter flattenconverter; - shuffle_test_helper(graph_IR, &flattenconverter, {1, 2, 3}); -} - -TEST(Converters, ATenFlattenStartEnddimConvertsCorrectly) { - // aten::flatten.using_ints(Tensor(a) self, int start_dim=0, int end_dim=-1) -> Tensor(a) - const auto graph_IR = gen_double_int_graph("flatten", "1", "2"); - baidu::mirana::poros::FlattenConverter flattenconverter; - shuffle_test_helper(graph_IR, &flattenconverter, {1, 2, 3}); -} - -TEST(Converters, ATenT1DConvertsCorrectly) { - // aten::t(Tensor(a) self) -> Tensor(a) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : Tensor = aten::t(%0) - %2 : Tensor = aten::relu(%1) - return (%2))IR"; - baidu::mirana::poros::AtenTConverter atentConverter; - shuffle_test_helper(graph_IR, &atentConverter, {5}); -} - -TEST(Converters, ATenT2DConvertsCorrectly) { - // aten::t(Tensor(a) self) -> Tensor(a) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : Tensor = aten::t(%0) - return (%1))IR"; - baidu::mirana::poros::AtenTConverter atentConverter; - shuffle_test_helper(graph_IR, &atentConverter, {5, 6}); -} - -TEST(Converters, ATenPixelShuffleConvertsCorrectly) { - // aten::pixel_shuffle(Tensor self, int upscale_factor) -> Tensor - const auto graph_IR = gen_pixel_shuffle_graph("3"); - baidu::mirana::poros::PixelShuffleConverter pixelshuffleconverter; - shuffle_test_helper(graph_IR, &pixelshuffleconverter, {1, 9, 4, 4}); -} - -TEST(Converters, ATenPixelShuffle3DConvertsCorrectly) { - // aten::pixel_shuffle(Tensor self, int upscale_factor) -> Tensor - const auto graph_IR = gen_pixel_shuffle_graph("3"); - baidu::mirana::poros::PixelShuffleConverter pixelshuffleconverter; - shuffle_test_helper(graph_IR, &pixelshuffleconverter, {9, 5, 6}); -} - -TEST(Converters, ATenPixelShuffle5DConvertsCorrectly) { - // aten::pixel_shuffle(Tensor self, int upscale_factor) -> Tensor - const auto graph_IR = gen_pixel_shuffle_graph("3"); - baidu::mirana::poros::PixelShuffleConverter pixelshuffleconverter; - shuffle_test_helper(graph_IR, &pixelshuffleconverter, {7, 8, 9, 5, 6}); -} - -static void shuffle_dynamic_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - const std::vector& input_data, - bool is_dynamic = false, - std::vector>* prewarm_data = nullptr) { - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = is_dynamic; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output, prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenViewdynamicConvertsCorrectly) { - // aten::view(Tensor(a) self, int[] size) -> Tensor(a) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=0]() - %2 : int = prim::Constant[value=1]() - %3 : int = aten::size(%0, %1) - %4 : int = aten::size(%0, %2) - %5 : int[] = prim::ListConstruct(%4, %3) - %6 : Tensor = aten::view(%0, %5) - return (%6))IR"; - baidu::mirana::poros::PermuteViewConverter permuteviewconverter; - std::vector input_data; - input_data.push_back(at::randn({2, 3}, {at::kCUDA})); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({4, 5}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({2, 3}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({2, 3}, {at::kCUDA})); - - shuffle_dynamic_test_helper(graph_IR, &permuteviewconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenReshapedynamicConvertsCorrectly) { - // aten::reshape(Tensor(a) self, int[] shape) -> Tensor(a) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : int[] = aten::size(%0) - %2 : int, %3 : int = prim::ListUnpack(%1) - %4 : int[] = prim::ListConstruct(%3, %2) - %5 : Tensor = aten::reshape(%0, %4) - return (%5))IR"; - baidu::mirana::poros::ReshapeConverter reshapeconverter; - std::vector input_data; - input_data.push_back(at::randn({2, 3}, {at::kCUDA})); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({4, 5}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({2, 3}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({2, 3}, {at::kCUDA})); - - shuffle_dynamic_test_helper(graph_IR, &reshapeconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenFlattenConvertsDynamicCorrectly) { - // aten::flatten.using_ints(Tensor(a) self, int start_dim=0, int end_dim=-1) -> Tensor(a) - const auto graph_IR = gen_double_int_graph("flatten", "0", "2"); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({10, 64, 128}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({5, 32, 64}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({5, 32, 64}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({5, 32, 64}, {at::kCUDA})); - baidu::mirana::poros::FlattenConverter flattenconverter; - shuffle_dy_test_helper(graph_IR, input_data, &flattenconverter, true, &prewarm_data); -} - -TEST(Converters, ATenFlattenConvertsDynamicNegStartEndCorrectly) { - // aten::flatten.using_ints(Tensor(a) self, int start_dim=0, int end_dim=-1) -> Tensor(a) - const auto graph_IR = gen_double_int_graph("flatten", "-3", "-2"); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({10, 64, 128, 32}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({5, 32, 64, 16}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({5, 32, 64, 16}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({5, 32, 64, 16}, {at::kCUDA})); - baidu::mirana::poros::FlattenConverter flattenconverter; - shuffle_dy_test_helper(graph_IR, input_data, &flattenconverter, true, &prewarm_data); -} - -TEST(Converters, ATenFlattenConvertsDynamicStartEqualEndCorrectly) { - // aten::flatten.using_ints(Tensor(a) self, int start_dim=0, int end_dim=-1) -> Tensor(a) - const auto graph_IR = gen_double_int_graph("flatten", "1", "1"); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({10, 64, 128, 32}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({5, 32, 64, 16}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({5, 32, 64, 16}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({5, 32, 64, 16}, {at::kCUDA})); - baidu::mirana::poros::FlattenConverter flattenconverter; - shuffle_dy_test_helper(graph_IR, input_data, &flattenconverter, true, &prewarm_data); -} \ No newline at end of file diff --git a/poros/unittest/converter/softmax_test.cpp b/poros/unittest/converter/softmax_test.cpp deleted file mode 100644 index 70f36a3ec1f..00000000000 --- a/poros/unittest/converter/softmax_test.cpp +++ /dev/null @@ -1,148 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file softmax_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/softmax.h" -#include "poros/util/test_util.h" - -static void softmax_test_helper(const std::string& graph_IR, - std::vector shape = {5}){ - std::vector input_data; - input_data.push_back(at::randn(shape, {at::kCUDA})); - // input_data.push_back(at::randint(0, 5, {5}, {at::kCUDA})); - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - baidu::mirana::poros::SoftmaxConverter softmaxconverter; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &softmaxconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -static std::string gen_softmax_graph(const std::string& dim) { - return R"IR( - graph(%0 : Tensor): - %1 : None = prim::Constant() - %2 : int = prim::Constant[value=)IR" + dim + R"IR(]() - %3 : Tensor = aten::softmax(%0, %2, %1) - return (%3))IR"; -} - -TEST(Converters, ATenSoftmax1DConvertsCorrectly) { - // aten::softmax.int(Tensor self, int dim, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_softmax_graph("0"); - softmax_test_helper(graph_IR, {5}); -} - -TEST(Converters, ATenSoftmaxNDConvertsCorrectlySub3DIndex) { - // aten::softmax.int(Tensor self, int dim, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_softmax_graph("1"); - softmax_test_helper(graph_IR, {1, 2, 3, 4, 5}); -} - -TEST(Converters, ATenSoftmaxNDConvertsCorrectlyAbove3DIndex) { - // aten::softmax.int(Tensor self, int dim, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_softmax_graph("3"); - softmax_test_helper(graph_IR, {1, 2, 3, 4, 5}); -} - -TEST(Converters, ATenSoftmaxNDConvertsCorrectlyNegtiveOneIndex) { - // aten::softmax.int(Tensor self, int dim, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_softmax_graph("-1"); - softmax_test_helper(graph_IR, {1, 2, 3, 4, 5}); -} - -TEST(Converters, ATenSoftmaxNDConvertsCorrectlyNegtiveIndex) { - // aten::softmax.int(Tensor self, int dim, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_softmax_graph("-2"); - softmax_test_helper(graph_IR, {1, 2, 3, 4, 5}); -} - -static void softmax_dy_test_helper(const std::string& graph_IR, - const std::vector& input_data, - bool is_dynamic = false, - std::vector>* prewarm_data = nullptr) { - baidu::mirana::poros::SoftmaxConverter softmaxconverter; - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = is_dynamic; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &softmaxconverter, - input_data, graph_output, poros_output, prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenSoftmaxInputSingleDimDynamicConvertsCorrectly) { - // aten::softmax.int(Tensor self, int dim, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_softmax_graph("0"); - - std::vector> prewarm_data = {{}, {}, {}}; - - prewarm_data[0].push_back(at::randn({60}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({40}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({40}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({40}, {at::kCUDA})); - - softmax_dy_test_helper(graph_IR, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenSoftmaxDynamicConvertsCorrectly) { - // aten::softmax.int(Tensor self, int dim, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_softmax_graph("2"); - - std::vector> prewarm_data = {{}, {}, {}}; - - prewarm_data[0].push_back(at::randn({20, 30, 40, 50}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({10, 20, 30, 40}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({10, 20, 30, 40}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({10, 20, 30, 40}, {at::kCUDA})); - - softmax_dy_test_helper(graph_IR, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenSoftmaxDynamicNegtiveDimConvertsCorrectly) { - // aten::softmax.int(Tensor self, int dim, ScalarType? dtype=None) -> Tensor - const auto graph_IR = gen_softmax_graph("-2"); - - std::vector> prewarm_data = {{}, {}, {}}; - - prewarm_data[0].push_back(at::randn({20, 30, 40, 50}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({10, 20, 30, 40}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({10, 20, 30, 40}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({10, 20, 30, 40}, {at::kCUDA})); - - softmax_dy_test_helper(graph_IR, input_data, true, &prewarm_data); -} \ No newline at end of file diff --git a/poros/unittest/converter/squeeze_test.cpp b/poros/unittest/converter/squeeze_test.cpp deleted file mode 100644 index 31d060d5496..00000000000 --- a/poros/unittest/converter/squeeze_test.cpp +++ /dev/null @@ -1,218 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file squeeze_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/squeeze.h" -#include "poros/util/test_util.h" - -static void squeeze_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - std::vector shape){ - std::vector input_data; - input_data.push_back(at::randn(shape, {at::kCUDA})); - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -static std::string gen_squeeze_one_input_schema_graph(const std::string& op) { - return R"IR( - graph(%0 : Tensor): - %2 : Tensor = aten::)IR" + op + R"IR((%0) - %3 : Tensor = aten::relu(%2) - return (%3))IR"; -} - -TEST(Converters, ATenSqueezeOneInputConvertsCorrectly) { - // aten::squeeze.dim(Tensor(a) self, int dim) -> Tensor(a) - const auto graph_IR = gen_squeeze_one_input_schema_graph("squeeze"); - baidu::mirana::poros::SqueezeConverter squeezeconverter; - squeeze_test_helper(graph_IR, &squeezeconverter, {4, 1, 3}); - squeeze_test_helper(graph_IR, &squeezeconverter, {4, 1, 1, 5}); -} - -static std::string gen_squeeze_graph(const std::string& op, const std::string& dim) { - return R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=)IR" + dim + R"IR(]() - %2 : Tensor = aten::)IR" + op + R"IR((%0, %1) - %3 : Tensor = aten::relu(%2) - return (%3))IR"; -} - -TEST(Converters, ATenSqueezeConvertsCorrectly) { - // aten::squeeze.dim(Tensor(a) self, int dim) -> Tensor(a) - const auto graph_IR = gen_squeeze_graph("squeeze", "1"); - baidu::mirana::poros::SqueezeConverter squeezeconverter; - squeeze_test_helper(graph_IR, &squeezeconverter, {4, 1, 3}); - squeeze_test_helper(graph_IR, &squeezeconverter, {4, 2, 3}); -} - -TEST(Converters, ATenSqueezeNegtiveConvertsCorrectly) { - // aten::squeeze.dim(Tensor(a) self, int dim) -> Tensor(a) - const auto graph_IR = gen_squeeze_graph("squeeze", "-1"); - baidu::mirana::poros::SqueezeConverter squeezeconverter; - squeeze_test_helper(graph_IR, &squeezeconverter, {4, 3, 1}); - squeeze_test_helper(graph_IR, &squeezeconverter, {4, 2, 3}); -} - -TEST(Converters, ATenUnSqueezeConvertsCorrectly) { - // aten::unsqueeze(Tensor(a) self, int dim) -> Tensor(a) - const auto graph_IR = gen_squeeze_graph("unsqueeze", "1"); - baidu::mirana::poros::UnSqueezeConverter unsqueezeconverter; - squeeze_test_helper(graph_IR, &unsqueezeconverter, {4, 3, 2}); -} - -TEST(Converters, ATenUnSqueezeNegtiveConvertsCorrectly) { - // aten::unsqueeze(Tensor(a) self, int dim) -> Tensor(a) - const auto graph_IR = gen_squeeze_graph("unsqueeze", "-1"); - baidu::mirana::poros::UnSqueezeConverter unsqueezeconverter; - squeeze_test_helper(graph_IR, &unsqueezeconverter, {4, 3, 2}); -} - -static void squeeze_dy_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - const std::vector& input_data, - bool is_dynamic = false, - std::vector>* prewarm_data = nullptr) { - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = is_dynamic; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output, prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -TEST(Converters, ATenSqueezeOneInputDynamicConvertsCorrectly) { - // aten::squeeze(Tensor(a) self, int dim) -> Tensor(a) - const auto graph_IR = gen_squeeze_one_input_schema_graph("squeeze"); - baidu::mirana::poros::SqueezeConverter squeezeconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({40, 1, 1, 60}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({20, 1, 1, 40}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({20, 1, 1, 40}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({20, 1, 1, 40}, {at::kCUDA})); - - squeeze_dy_test_helper(graph_IR, &squeezeconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenUnSqueezeDynamicConvertsCorrectly) { - // aten::unsqueeze(Tensor(a) self, int dim) -> Tensor(a) - const auto graph_IR = gen_squeeze_graph("unsqueeze", "2"); - baidu::mirana::poros::UnSqueezeConverter unsqueezeconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - - prewarm_data[0].push_back(at::randn({40, 50, 60}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({20, 30, 40}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({20, 30, 40}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({20, 30, 40}, {at::kCUDA})); - - squeeze_dy_test_helper(graph_IR, &unsqueezeconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenUnSqueezeInputSingleDimDynamicConvertsCorrectly) { - // aten::unsqueeze(Tensor(a) self, int dim) -> Tensor(a) - const auto graph_IR = gen_squeeze_graph("unsqueeze", "0"); - baidu::mirana::poros::UnSqueezeConverter unsqueezeconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - - prewarm_data[0].push_back(at::randn({40}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({20}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({20}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({20}, {at::kCUDA})); - - squeeze_dy_test_helper(graph_IR, &unsqueezeconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenUnSqueezeDynamicNegtiveDimConvertsCorrectly) { - // aten::unsqueeze(Tensor(a) self, int dim) -> Tensor(a) - const auto graph_IR = gen_squeeze_graph("unsqueeze", "-1"); - baidu::mirana::poros::UnSqueezeConverter unsqueezeconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - - prewarm_data[0].push_back(at::randn({40, 50, 60}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({20, 30, 40}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({20, 30, 40}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({20, 30, 40}, {at::kCUDA})); - - squeeze_dy_test_helper(graph_IR, &unsqueezeconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenSqueezeDynamicConvertsCorrectly) { - // aten::squeeze(Tensor(a) self, int dim) -> Tensor(a) - const auto graph_IR = gen_squeeze_graph("squeeze", "1"); - baidu::mirana::poros::SqueezeConverter squeezeconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - - prewarm_data[0].push_back(at::randn({40, 1, 60}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({20, 1, 40}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({20, 1, 40}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({20, 1, 40}, {at::kCUDA})); - - squeeze_dy_test_helper(graph_IR, &squeezeconverter, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenSqueezeDynamicNegtiveDimConvertsCorrectly) { - // aten::squeeze(Tensor(a) self, int dim) -> Tensor(a) - const auto graph_IR = gen_squeeze_graph("squeeze", "-1"); - baidu::mirana::poros::SqueezeConverter squeezeconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - - prewarm_data[0].push_back(at::randn({1, 60, 1}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({1, 40, 1}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({1, 40, 1}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({1, 40, 1}, {at::kCUDA})); - - squeeze_dy_test_helper(graph_IR, &squeezeconverter, input_data, true, &prewarm_data); -} \ No newline at end of file diff --git a/poros/unittest/converter/stack_test.cpp b/poros/unittest/converter/stack_test.cpp deleted file mode 100644 index dafccec4266..00000000000 --- a/poros/unittest/converter/stack_test.cpp +++ /dev/null @@ -1,186 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file stack_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/stack.h" -#include "poros/util/test_util.h" - -static void stack_test_helper(const std::string& graph_IR, - std::vector shape1 = {5}, - std::vector shape2 = {5}, - bool Triple_inputs = false, - std::vector shape3 = {5}){ - std::vector input_data; - input_data.push_back(at::randn(shape1, {at::kCUDA})); - input_data.push_back(at::randn(shape2, {at::kCUDA})); - if (Triple_inputs){ - input_data.push_back(at::randn(shape3, {at::kCUDA})); - } - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - baidu::mirana::poros::StackConverter stackconverter; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &stackconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -static std::string gen_double_inputs_stack_graph(const std::string& dim) { - return R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : Tensor[] = prim::ListConstruct(%0, %1) - %3 : int = prim::Constant[value=)IR" + dim + R"IR(]() - %4 : Tensor = aten::stack(%2, %3) - return (%4))IR"; -} - -static std::string gen_triple_inputs_stack_graph(const std::string& dim) { - return R"IR( - graph(%0 : Tensor, %1 : Tensor, %2 : Tensor): - %3 : Tensor[] = prim::ListConstruct(%0, %1, %2) - %4 : int = prim::Constant[value=)IR" + dim + R"IR(]() - %5 : Tensor = aten::stack(%3, %4) - return (%5))IR"; -} - -TEST(Converters, ATenStackDoubleTensorConvertsCorrectly) { - // aten::stack(Tensor[] tensors, int dim=0) -> Tensor - const auto graph_IR = gen_double_inputs_stack_graph("0"); - stack_test_helper(graph_IR); -} - -TEST(Converters, ATenStackDoubleTensoroneDimConvertsCorrectly) { - // aten::stack(Tensor[] tensors, int dim=0) -> Tensor - const auto graph_IR = gen_double_inputs_stack_graph("1"); - stack_test_helper(graph_IR, {5, 3}, {5, 3}); -} - -TEST(Converters, ATenStackTripleTensorConvertsCorrectly) { - // aten::stack(Tensor[] tensors, int dim=0) -> Tensor - const auto graph_IR = gen_triple_inputs_stack_graph("2"); - stack_test_helper(graph_IR, {5, 2, 3}, {5, 2, 3}, true, {5, 2, 3}); -} - -TEST(Converters, ATenVstackDoubleTensorConvertsCorrectly) { - // aten::vstack(Tensor[] tensors) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : Tensor[] = prim::ListConstruct(%0, %1) - %3 : Tensor = aten::vstack(%2) - return (%3))IR"; - stack_test_helper(graph_IR, {3, 1}, {3, 1}); -} - -TEST(Converters, ATenVstackTripleTensorConvertsCorrectly) { - // aten::vstack(Tensor[] tensors) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor, %2 : Tensor): - %3 : Tensor[] = prim::ListConstruct(%0, %1, %2) - %4 : Tensor = aten::vstack(%3) - return (%4))IR"; - stack_test_helper(graph_IR, {5, 2, 3}, {5, 2, 3}, true, {5, 2, 3}); -} - - -static void stack_dy_test_helper(const std::string& graph_IR, - const std::vector& input_data, - bool is_dynamic = false, - std::vector>* prewarm_data = nullptr) { - baidu::mirana::poros::StackConverter stackconverter; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = is_dynamic; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &stackconverter, - input_data, graph_output, poros_output, prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenStackDoubleTensorDynamicTestConvertsCorrectly) { - // aten::stack(Tensor[] tensors, int dim=0) -> Tensor - const auto graph_IR = gen_double_inputs_stack_graph("2"); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({10, 5, 3, 3}, {at::kCUDA})); - prewarm_data[0].push_back(at::randn({10, 5, 3, 3}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({5, 5, 3, 3}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({5, 5, 3, 3}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({5, 5, 3, 3}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({5, 5, 3, 3}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({5, 5, 3, 3}, {at::kCUDA})); - input_data.push_back(at::randn({5, 5, 3, 3}, {at::kCUDA})); - - stack_dy_test_helper(graph_IR, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenStackDoubleTensorDynamicNegDimTestConvertsCorrectly) { - // aten::stack(Tensor[] tensors, int dim=0) -> Tensor - const auto graph_IR = gen_double_inputs_stack_graph("-2"); - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({10, 5, 3, 3}, {at::kCUDA})); - prewarm_data[0].push_back(at::randn({10, 5, 3, 3}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({5, 5, 3, 3}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({5, 5, 3, 3}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({5, 5, 3, 3}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({5, 5, 3, 3}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({5, 5, 3, 3}, {at::kCUDA})); - input_data.push_back(at::randn({5, 5, 3, 3}, {at::kCUDA})); - - stack_dy_test_helper(graph_IR, input_data, true, &prewarm_data); -} - -TEST(Converters, ATenVStackDoubleTensorDynamicTestConvertsCorrectly) { - // aten::vstack(Tensor[] tensors) -> Tensor - const auto graph_IR = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %3 : Tensor[] = prim::ListConstruct(%0, %1) - %4 : Tensor = aten::vstack(%3) - return (%4))IR"; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({10, 5, 3, 3}, {at::kCUDA})); - prewarm_data[0].push_back(at::randn({10, 5, 3, 3}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({5, 5, 3, 3}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({5, 5, 3, 3}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({5, 5, 3, 3}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({5, 5, 3, 3}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::randn({5, 5, 3, 3}, {at::kCUDA})); - input_data.push_back(at::randn({5, 5, 3, 3}, {at::kCUDA})); - - stack_dy_test_helper(graph_IR, input_data, true, &prewarm_data); -} \ No newline at end of file diff --git a/poros/unittest/converter/to_test.cpp b/poros/unittest/converter/to_test.cpp deleted file mode 100644 index ce643cbe96f..00000000000 --- a/poros/unittest/converter/to_test.cpp +++ /dev/null @@ -1,69 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file to_test.cpp -* @author wangrui39@baidu.com -* @date Sunday November 14 11:36:11 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/to.h" -#include "poros/util/test_util.h" - -static void add_test_helper(const std::string& graph_IR, - baidu::mirana::poros::IConverter* converter, - std::vector shape1 = {5}, - std::vector shape2 = {5}){ - std::vector input_data; - - input_data.push_back(at::ones(shape1, {at::kCUDA})); - input_data.push_back(at::ones(shape2, {at::kCUDA})); - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, converter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); -} - -static std::string gen_to_graph() { - std::string graph = R"IR( - graph(%0 : Tensor, %1 : Tensor): - %2 : float = prim::Constant[value=2]() - %3 : int = prim::Constant[value=3]() - %4 : bool = prim::Constant[value=0]() - %5 : None = prim::Constant() - %6 : Tensor = aten::to(%0, %3, %4, %4, %5) - %7 : Tensor = aten::to(%1, %6, %4, %4, %5) - %35 : Device = prim::Constant[value="cuda"]() - %6 : Tensor = aten::to(%6, %35, %3, %4, %4, %5) - %7 : Tensor = aten::to(%7, %35, %3, %4, %4, %5) - %8 : int = prim::Constant[value=1]() - %9 : Tensor = aten::add(%6, %7, %8) - return (%9))IR"; - return graph; -} - -TEST(Converters, ATenToConvertsCorrectly) { - const auto graph_IR = gen_to_graph(); - baidu::mirana::poros::ToConverter toconverter; - add_test_helper(graph_IR, &toconverter, {3, 4}, {3, 4}); -} \ No newline at end of file diff --git a/poros/unittest/converter/topk_test.cpp b/poros/unittest/converter/topk_test.cpp deleted file mode 100644 index bee9f473f5d..00000000000 --- a/poros/unittest/converter/topk_test.cpp +++ /dev/null @@ -1,85 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file topk_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/topk.h" -#include "poros/util/test_util.h" - -static void topk_test_helper(const std::string& graph_IR, - std::vector shape) { - std::vector input_data; - input_data.push_back(at::randn(shape, {at::kCUDA})); - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - baidu::mirana::poros::TopkConverter topkconverter; - - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &topkconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(2, graph_output.size()); - ASSERT_EQ(2, poros_output.size()); - - // ASSERT_TRUE(baidu::mirana::poros::testutil::almostEqual(graph_output[0], poros_output[0], 2e-6)); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); - ASSERT_TRUE(graph_output[1].equal(poros_output[1])); -} -static std::string gen_topk_graph(const std::string& k, - const std::string& dim, - const std::string& largest, - const std::string& sorted) { - return R"IR( - graph(%0 : Tensor): - %1 : int = prim::Constant[value=)IR" + k + R"IR(]() - %2 : int = prim::Constant[value=)IR" + dim + R"IR(]() - %3 : bool = prim::Constant[value=)IR" + largest + R"IR(]() - %4 : bool = prim::Constant[value=)IR" + sorted + R"IR(]() - %5 : Tensor, %6 : Tensor = aten::topk(%0, %1, %2, %3, %4) - return (%5, %6))IR"; -} - -TEST(Converters, ATenTopkConvertsCorrectly) { - // aten::topk(Tensor self, int k, int dim=-1, bool largest=True, bool sorted=True) -> (Tensor values, Tensor indices) - const auto graph_IR = gen_topk_graph("10", "0", "1", "1"); - topk_test_helper(graph_IR, {20, 10}); -} - -TEST(Converters, ATenTopkDimConvertsCorrectly) { - // aten::topk(Tensor self, int k, int dim=-1, bool largest=True, bool sorted=True) -> (Tensor values, Tensor indices) - const auto graph_IR = gen_topk_graph("5", "1", "1", "1"); - topk_test_helper(graph_IR, {20, 10}); -} - -TEST(Converters, ATenTopkDimNegtiveConvertsCorrectly) { - // aten::topk(Tensor self, int k, int dim=-1, bool largest=True, bool sorted=True) -> (Tensor values, Tensor indices) - const auto graph_IR = gen_topk_graph("5", "-1", "1", "1"); - topk_test_helper(graph_IR, {20, 10}); -} - -TEST(Converters, ATenTopklargestConvertsCorrectly) { - // aten::topk(Tensor self, int k, int dim=-1, bool largest=True, bool sorted=True) -> (Tensor values, Tensor indices) - const auto graph_IR = gen_topk_graph("10", "0", "0", "1"); - topk_test_helper(graph_IR, {20, 10}); -} - -// sorted argument is not used in TensorRT for aten::topk \ No newline at end of file diff --git a/poros/unittest/converter/unary_test.cpp b/poros/unittest/converter/unary_test.cpp deleted file mode 100644 index 6bda6ea8e8a..00000000000 --- a/poros/unittest/converter/unary_test.cpp +++ /dev/null @@ -1,216 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file unary_test.cpp -* @author tianshaoqing@baidu.com -* @date Wed Sep 27 11:24:21 CST 2021 -* @brief -**/ -#include -#include - -#include "poros/converter/gpu/unary.h" -#include "poros/util/test_util.h" - -static void unary_test_helper(const std::string& op, - std::vector shape = {10}){ - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %1 : Tensor = aten::)IR" +op + R"IR((%0) - return (%1))IR"; - std::vector input_data; - float offset = 0; - if(op == "acosh"){ - offset += 1; - } - if(op == "abs" || op == "neg"){ - offset -= 0.5; - } - auto input_tensor = at::empty(shape, {at::kCUDA}).uniform_(0 + offset, 0.5 + offset); - if(op == "round") { - input_tensor = input_tensor * 50; - } - input_data.push_back(input_tensor); - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - baidu::mirana::poros::UnaryConverter unaryconverter; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &unaryconverter, - input_data, graph_output, poros_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - - ASSERT_TRUE(baidu::mirana::poros::testutil::almost_equal(graph_output[0], poros_output[0], 2e-6)); - // ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} - -TEST(Converters, ATenCosConvertsCorrectly) { - // aten::cos(Tensor self) -> Tensor - unary_test_helper("cos"); -} - -TEST(Converters, ATenAcosConvertsCorrectly) { - // aten::acos(Tensor self) -> Tensor - unary_test_helper("acos"); -} - -TEST(Converters, ATenCoshConvertsCorrectly) { - // aten::cosh(Tensor self) -> Tensor - unary_test_helper("cosh"); -} - -TEST(Converters, ATenSinConvertsCorrectly) { - // aten::sin(Tensor self) -> Tensor - unary_test_helper("sin"); -} - -TEST(Converters, ATenAsinConvertsCorrectly) { - // aten::asin(Tensor self) -> Tensor - unary_test_helper("asin"); -} - -TEST(Converters, ATenSinhConvertsCorrectly) { - // aten::sinh(Tensor self) -> Tensor - unary_test_helper("sinh"); -} - -TEST(Converters, ATenTanConvertsCorrectly) { - // aten::tan(Tensor self) -> Tensor - unary_test_helper("tan"); -} - -TEST(Converters, ATenAtanConvertsCorrectly) { - // aten::atan(Tensor self) -> Tensor - unary_test_helper("atan"); -} - -TEST(Converters, ATenAbsConvertsCorrectly) { - // aten::abs(Tensor self) -> Tensor - unary_test_helper("abs"); -} - -TEST(Converters, ATenFloorConvertsCorrectly) { - // aten::floor(Tensor self) -> Tensor - unary_test_helper("floor"); -} - -TEST(Converters, ATenReciprocalConvertsCorrectly) { - // aten::reciprocal(Tensor self) -> Tensor - unary_test_helper("reciprocal"); -} - -TEST(Converters, ATenLogConvertsCorrectly) { - // aten::log(Tensor self) -> Tensor - unary_test_helper("log"); -} - -TEST(Converters, ATenCeilConvertsCorrectly) { - // aten::ceil(Tensor self) -> Tensor - unary_test_helper("ceil"); -} - -TEST(Converters, ATenSqrtConvertsCorrectly) { - // aten::sqrt(Tensor self) -> Tensor - unary_test_helper("sqrt"); -} - -TEST(Converters, ATenExpConvertsCorrectly) { - // aten::exp(Tensor self) -> Tensor - unary_test_helper("exp"); -} - -TEST(Converters, ATenNegConvertsCorrectly) { - // aten::neg(Tensor self) -> Tensor - unary_test_helper("neg"); -} - -TEST(Converters, ATenErfConvertsCorrectly) { - // aten::erf(Tensor self) -> Tensor - unary_test_helper("erf"); -} - -TEST(Converters, ATenAsinhConvertsCorrectly) { - // aten::asinh(Tensor self) -> Tensor - unary_test_helper("asinh"); -} - -TEST(Converters, ATenAcoshConvertsCorrectly) { - // aten::acosh(Tensor self) -> Tensor - unary_test_helper("acosh"); -} - -TEST(Converters, ATenAtanhConvertsCorrectly) { - // aten::atanh(Tensor self) -> Tensor - unary_test_helper("atanh"); -} - -TEST(Converters, ATenLog2ConvertsCorrectly) { - // aten::log2(Tensor self) -> Tensor - unary_test_helper("log2"); -} - -TEST(Converters, ATenLog10ConvertsCorrectly) { - // aten::log10(Tensor self) -> Tensor - unary_test_helper("log10"); -} - -TEST(Converters, ATenRoundConvertsCorrectly) { - // aten::round(Tensor self) -> (Tensor) - unary_test_helper("round"); -} - -TEST(Converters, ATenFloorFloat2IntConvertsCorrectly) { - // aten::floor.float(float a) -> (int) - const auto graph_IR = R"IR( - graph(%0 : Tensor): - %dim0 : int = prim::Constant[value=0]() - %dim1 : int = prim::Constant[value=1]() - %1 : float = prim::Constant[value=-1.5]() - %2 : int = aten::size(%0, %dim0) - %3 : int = aten::size(%0, %dim1) - %4 : float = aten::div(%2, %3) - %5 : int = aten::floor(%4) - %6 : int = aten::floor(%1) - %7 : int[] = prim::ListConstruct(%5, %6) - %8 : NoneType = prim::Constant() - %9 : bool = prim::Constant[value=0]() - %10 : Device = prim::Constant[value="cuda:0"]() - %11 : Tensor = aten::tensor(%7, %8, %10, %9) - return (%11))IR"; - - baidu::mirana::poros::UnaryConverter unaryconverter; - - std::vector> prewarm_data = {{}, {}, {}}; - prewarm_data[0].push_back(at::randn({7, 2}, {at::kCUDA})); - prewarm_data[1].push_back(at::randn({3, 2}, {at::kCUDA})); - prewarm_data[2].push_back(at::randn({5, 2}, {at::kCUDA})); - - std::vector input_data; - input_data.push_back(at::ones({7, 2}, {at::kCUDA})); - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - poros_option.is_dynamic = true; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector poros_output; - ASSERT_TRUE(baidu::mirana::poros::testutil::run_graph_and_poros(graph_IR, poros_option, &unaryconverter, - input_data, graph_output, poros_output, &prewarm_data)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, poros_output.size()); - ASSERT_TRUE(graph_output[0].equal(poros_output[0])); -} \ No newline at end of file diff --git a/poros/unittest/op_fuser/fuse_conv_bn_test.cpp b/poros/unittest/op_fuser/fuse_conv_bn_test.cpp deleted file mode 100644 index 0416b9308a1..00000000000 --- a/poros/unittest/op_fuser/fuse_conv_bn_test.cpp +++ /dev/null @@ -1,233 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file fuse_conv_bn_test.cpp -* @author Lin Xiao Chun (linxiaochun@baidu.com) -* @date 2022-03-31 16:11:18 -* @brief -**/ - -#include -#include - -#include "poros/lowering/fuse_conv_bn.h" -#include "poros/lowering/op_fuse_pass.h" -#include "poros/util/graph_test_helper.h" - -std::vector ones(size_t n) { - return std::vector(n, 1); -} - -std::vector zeros(size_t n) { - return std::vector(n, 0); -} - -static void fuse_test_helper(const std::string &graph_IR, - const size_t &dim, - std::shared_ptr fuser, - std::vector input_shape, - std::vector conv_w_shape, - std::vector conv_b_shape -) { - std::vector input_data; - input_data.push_back(at::randn(input_shape, {at::kCPU})); - input_data.push_back(at::randn(conv_w_shape, {at::kCPU})); - input_data.push_back(at::randn(conv_b_shape, {at::kCPU})); - - input_data.push_back(at::IntArrayRef(ones(dim))); //stride - input_data.push_back(at::IntArrayRef(zeros(dim))); //padding - input_data.push_back(at::IntArrayRef(ones(dim))); //dilation - - auto bn_shape = conv_b_shape; - input_data.push_back(at::randn(bn_shape, {at::kCPU})); //weight - input_data.push_back(at::randn(bn_shape, {at::kCPU})); //bias - input_data.push_back(at::randn(bn_shape, {at::kCPU})); //mean - input_data.push_back(at::abs(at::randn(bn_shape, {at::kCPU}))); //var - const std::vector input_data_type_mask = { - baidu::mirana::poros::graphtester::InputTensor, - - baidu::mirana::poros::graphtester::ConstantTensor, - baidu::mirana::poros::graphtester::ConstantTensor, - - baidu::mirana::poros::graphtester::ConstantIntVector, - baidu::mirana::poros::graphtester::ConstantIntVector, - baidu::mirana::poros::graphtester::ConstantIntVector, - - baidu::mirana::poros::graphtester::ConstantTensor, - baidu::mirana::poros::graphtester::ConstantTensor, - baidu::mirana::poros::graphtester::ConstantTensor, - baidu::mirana::poros::graphtester::ConstantTensor, - - }; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector fused_output; - ASSERT_TRUE(baidu::mirana::poros::graphtester::run_graph_and_fused_graph(graph_IR, poros_option, fuser, - input_data, input_data_type_mask, - graph_output, fused_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, fused_output.size()); - ASSERT_TRUE(baidu::mirana::poros::graphtester::almost_equal(graph_output[0], fused_output[0], 1e-6)); -} - -static std::string gen_conv3d_batch_norm3d_graph() { - - return R"IR( - graph(%x : Tensor, %conv_w : Tensor, %conv_b : Tensor, %conv_stride : Tensor, %conv_padding : Tensor, %conv_dilation : Tensor, %bn_w : Tensor, %bn_b : Tensor, %bn_m : Tensor, %bn_v : Tensor): - - %3 : int = prim::Constant[value=1]() - %conv_out : Tensor = aten::conv3d(%x, %conv_w, %conv_b, %conv_stride, %conv_padding, %conv_dilation, %3) - %7 : bool = prim::Constant[value=0]() - %4 : bool = prim::Constant[value=1]() - %8 : float = prim::Constant[value=1.0000000000000001e-05]() - %9 : float = prim::Constant[value=0.10000000000000001]() - %10 : Tensor = aten::batch_norm(%conv_out, %bn_w, %bn_b, %bn_m, %bn_v, %7, %9, %8, %4) - return (%10))IR"; -} - -static std::string gen_conv2d_batch_norm2d_graph() { - - return R"IR( - graph(%x : Tensor, %conv_w : Tensor, %conv_b : Tensor, %conv_stride : Tensor, %conv_padding : Tensor, %conv_dilation : Tensor, %bn_w : Tensor, %bn_b : Tensor, %bn_m : Tensor, %bn_v : Tensor): - - %3 : int = prim::Constant[value=1]() - %conv_out : Tensor = aten::conv2d(%x, %conv_w, %conv_b, %conv_stride, %conv_padding, %conv_dilation, %3) - %7 : bool = prim::Constant[value=0]() - %4 : bool = prim::Constant[value=1]() - %8 : float = prim::Constant[value=1.0000000000000001e-05]() - %9 : float = prim::Constant[value=0.10000000000000001]() - %10 : Tensor = aten::batch_norm(%conv_out, %bn_w, %bn_b, %bn_m, %bn_v, %7, %9, %8, %4) - return (%10))IR"; -} - -static std::string gen_conv1d_batch_norm1d_graph() { - - return R"IR( - graph(%x : Tensor, %conv_w : Tensor, %conv_b : Tensor, %conv_stride : Tensor, %conv_padding : Tensor, %conv_dilation : Tensor, %bn_w : Tensor, %bn_b : Tensor, %bn_m : Tensor, %bn_v : Tensor): - - %3 : int = prim::Constant[value=1]() - %conv_out : Tensor = aten::conv1d(%x, %conv_w, %conv_b, %conv_stride, %conv_padding, %conv_dilation, %3) - %7 : bool = prim::Constant[value=0]() - %4 : bool = prim::Constant[value=1]() - %8 : float = prim::Constant[value=1.0000000000000001e-05]() - %9 : float = prim::Constant[value=0.10000000000000001]() - %10 : Tensor = aten::batch_norm(%conv_out, %bn_w, %bn_b, %bn_m, %bn_v, %7, %9, %8, %4) - return (%10))IR"; -} - -static std::string gen_convolution_batch_norm3d_graph() { - - return R"IR( - graph(%x : Tensor, %conv_w : Tensor, %conv_b : Tensor, %conv_stride : Tensor, %conv_padding : Tensor, %conv_dilation : Tensor, %bn_w : Tensor, %bn_b : Tensor, %bn_m : Tensor, %bn_v : Tensor): - - %7 : bool = prim::Constant[value=0]() - %4 : bool = prim::Constant[value=1]() - %3 : int = prim::Constant[value=1]() - %output_padding : int[] = prim::Constant[value=[0, 0, 0]]() - %conv_out : Tensor = aten::_convolution(%x, %conv_w, %conv_b, %conv_stride, %conv_padding, %conv_dilation, %7, %output_padding, %3, %7, %7, %4, %4) - %8 : float = prim::Constant[value=1.0000000000000001e-05]() - %9 : float = prim::Constant[value=0.10000000000000001]() - %10 : Tensor = aten::batch_norm(%conv_out, %bn_w, %bn_b, %bn_m, %bn_v, %7, %9, %8, %4) - return (%10))IR"; -} - -static std::string gen_convolution_batch_norm2d_graph() { - - return R"IR( - graph(%x : Tensor, %conv_w : Tensor, %conv_b : Tensor, %conv_stride : Tensor, %conv_padding : Tensor, %conv_dilation : Tensor, %bn_w : Tensor, %bn_b : Tensor, %bn_m : Tensor, %bn_v : Tensor): - - %7 : bool = prim::Constant[value=0]() - %4 : bool = prim::Constant[value=1]() - %3 : int = prim::Constant[value=1]() - %output_padding : int[] = prim::Constant[value=[0, 0]]() - %conv_out : Tensor = aten::_convolution(%x, %conv_w, %conv_b, %conv_stride, %conv_padding, %conv_dilation, %7, %output_padding, %3, %7, %7, %4, %4) - %8 : float = prim::Constant[value=1.0000000000000001e-05]() - %9 : float = prim::Constant[value=0.10000000000000001]() - %10 : Tensor = aten::batch_norm(%conv_out, %bn_w, %bn_b, %bn_m, %bn_v, %7, %9, %8, %4) - return (%10))IR"; -} - -static std::string gen_convolution_batch_norm1d_graph() { - - return R"IR( - graph(%x : Tensor, %conv_w : Tensor, %conv_b : Tensor, %conv_stride : Tensor, %conv_padding : Tensor, %conv_dilation : Tensor, %bn_w : Tensor, %bn_b : Tensor, %bn_m : Tensor, %bn_v : Tensor): - - %7 : bool = prim::Constant[value=0]() - %4 : bool = prim::Constant[value=1]() - %3 : int = prim::Constant[value=1]() - %output_padding : int[] = prim::Constant[value=[0]]() - %conv_out : Tensor = aten::_convolution(%x, %conv_w, %conv_b, %conv_stride, %conv_padding, %conv_dilation, %7, %output_padding, %3, %7, %7, %4, %4) - %8 : float = prim::Constant[value=1.0000000000000001e-05]() - %9 : float = prim::Constant[value=0.10000000000000001]() - %10 : Tensor = aten::batch_norm(%conv_out, %bn_w, %bn_b, %bn_m, %bn_v, %7, %9, %8, %4) - return (%10))IR"; -} - -TEST(Fusers, ATenFuseConvBN3d_Test) { - const auto graph_IR = gen_conv3d_batch_norm3d_graph(); - auto fuser = std::make_shared(); - - fuse_test_helper(graph_IR, 3, fuser, {1, 2, 3, 4, 5}, {3, 2, 3, 3, 3}, {3}); - fuse_test_helper(graph_IR, 3, fuser, {3, 5, 8, 4, 6}, {12, 5, 3, 3, 3}, {12}); - fuse_test_helper(graph_IR, 3, fuser, {1, 2, 3, 4, 5}, {3, 2, 3, 3, 3}, {3}); -} - -TEST(Fusers, ATenFuseConvBN2d_Test) { - const auto graph_IR = gen_conv2d_batch_norm2d_graph(); - auto fuser = std::make_shared(); - - fuse_test_helper(graph_IR, 2, fuser, {1, 2, 4, 5}, {3, 2, 3, 3}, {3}); - fuse_test_helper(graph_IR, 2, fuser, {3, 5, 4, 6}, {12, 5, 3, 3}, {12}); - fuse_test_helper(graph_IR, 2, fuser, {1, 2, 4, 5}, {3, 2, 3, 3}, {3}); -} - -TEST(Fusers, ATenFuseConvBN1d_Test) { - const auto graph_IR = gen_conv1d_batch_norm1d_graph(); - auto fuser = std::make_shared(); - - fuse_test_helper(graph_IR, 1, fuser, {1, 2, 5}, {3, 2, 3}, {3}); - fuse_test_helper(graph_IR, 1, fuser, {3, 5, 6}, {12, 5, 3}, {12}); - fuse_test_helper(graph_IR, 1, fuser, {1, 2, 5}, {3, 2, 3}, {3}); -} - -TEST(Fusers, ATenFuseConvolutionBN3d_Test) { - const auto graph_IR = gen_convolution_batch_norm3d_graph(); - auto fuser = std::make_shared(); - - fuse_test_helper(graph_IR, 3, fuser, {1, 2, 3, 4, 5}, {3, 2, 3, 3, 3}, {3}); - fuse_test_helper(graph_IR, 3, fuser, {3, 5, 8, 4, 6}, {12, 5, 3, 3, 3}, {12}); - fuse_test_helper(graph_IR, 3, fuser, {1, 2, 3, 4, 5}, {3, 2, 3, 3, 3}, {3}); -} - -TEST(Fusers, ATenFuseConvolutionBN2d_Test) { - const auto graph_IR = gen_convolution_batch_norm2d_graph(); - auto fuser = std::make_shared(); - - fuse_test_helper(graph_IR, 2, fuser, {1, 2, 4, 5}, {3, 2, 3, 3}, {3}); - fuse_test_helper(graph_IR, 2, fuser, {3, 5, 4, 6}, {12, 5, 3, 3}, {12}); - fuse_test_helper(graph_IR, 2, fuser, {1, 2, 4, 5}, {3, 2, 3, 3}, {3}); -} - -TEST(Fusers, ATenFuseConvolutionBN1d_Test) { - const auto graph_IR = gen_convolution_batch_norm1d_graph(); - auto fuser = std::make_shared(); - - fuse_test_helper(graph_IR, 1, fuser, {1, 2, 5}, {3, 2, 3}, {3}); - fuse_test_helper(graph_IR, 1, fuser, {3, 5, 6}, {12, 5, 3}, {12}); - fuse_test_helper(graph_IR, 1, fuser, {1, 2, 5}, {3, 2, 3}, {3}); -} \ No newline at end of file diff --git a/poros/unittest/op_fuser/fuse_conv_mul_test.cpp b/poros/unittest/op_fuser/fuse_conv_mul_test.cpp deleted file mode 100644 index a74ccb28225..00000000000 --- a/poros/unittest/op_fuser/fuse_conv_mul_test.cpp +++ /dev/null @@ -1,137 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file: /icode-poros/baidu/mirana/poros/unnitest/op_fuser/fuse_conv_mul_test.cpp -* @author: zhangfan51@baidu.com -* @data: 2022-04-24 19:00:03 -* @brief: -**/ - -#include -#include - -#include "poros/lowering/fuse_conv_mul.h" -#include "poros/lowering/op_fuse_pass.h" -#include "poros/util/graph_test_helper.h" - -static std::vector ones(size_t n) { - return std::vector(n, 1); -} - -static std::vector zeros(size_t n) { - return std::vector(n, 0); -} - -static void fuse_test_helper(const std::string &graph_IR, - const size_t &dim, - std::shared_ptr fuser, - std::vector input_shape, - std::vector conv_w_shape, - std::vector conv_b_shape -) { - std::vector input_data; - input_data.push_back(at::randn(input_shape, {at::kCPU})); - input_data.push_back(at::randn(conv_w_shape, {at::kCPU})); - input_data.push_back(at::randn(conv_b_shape, {at::kCPU})); - - input_data.push_back(at::IntArrayRef(ones(dim))); //stride - input_data.push_back(at::IntArrayRef(zeros(dim))); //padding - input_data.push_back(at::IntArrayRef(ones(dim))); //dilation - - const std::vector input_data_type_mask = { - baidu::mirana::poros::graphtester::InputTensor, - - baidu::mirana::poros::graphtester::ConstantTensor, - baidu::mirana::poros::graphtester::ConstantTensor, - - baidu::mirana::poros::graphtester::ConstantIntVector, - baidu::mirana::poros::graphtester::ConstantIntVector, - baidu::mirana::poros::graphtester::ConstantIntVector, - }; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector fused_output; - ASSERT_TRUE(baidu::mirana::poros::graphtester::run_graph_and_fused_graph(graph_IR, poros_option, fuser, - input_data, input_data_type_mask, - graph_output, fused_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, fused_output.size()); - ASSERT_TRUE(baidu::mirana::poros::graphtester::almost_equal(graph_output[0], fused_output[0], 1e-6)); -} - -static std::string gen_conv3d_mul_graph() { - return R"IR( - graph(%x : Tensor, %conv_w : Tensor, %conv_b : Tensor, %conv_stride : Tensor, - %conv_padding : Tensor, %conv_dilation : Tensor): - %3 : int = prim::Constant[value=1]() - %4 : float = prim::Constant[value=2.0]() - %conv_out : Tensor = aten::conv3d(%x, %conv_w, %conv_b, %conv_stride, %conv_padding, %conv_dilation, %3) - %5 : Tensor = aten::mul(%conv_out, %4) - return (%5))IR"; -} - -static std::string gen_conv2d_mul_graph() { - - return R"IR( - graph(%x : Tensor, %conv_w : Tensor, %conv_b : Tensor, %conv_stride : Tensor, - %conv_padding : Tensor, %conv_dilation : Tensor): - %3 : int = prim::Constant[value=1]() - %4 : float = prim::Constant[value=2.0]() - %conv_out : Tensor = aten::conv2d(%x, %conv_w, %conv_b, %conv_stride, %conv_padding, %conv_dilation, %3) - %5 : Tensor = aten::mul(%conv_out, %4) - return (%5))IR"; -} - -static std::string gen_conv1d_mul_graph() { - - return R"IR( - graph(%x : Tensor, %conv_w : Tensor, %conv_b : Tensor, %conv_stride : Tensor, - %conv_padding : Tensor, %conv_dilation : Tensor): - %3 : int = prim::Constant[value=1]() - %4 : float = prim::Constant[value=2.0]() - %conv_out : Tensor = aten::conv1d(%x, %conv_w, %conv_b, %conv_stride, %conv_padding, %conv_dilation, %3) - %5 : Tensor = aten::mul(%conv_out, %4) - return (%5))IR"; -} - -TEST(Fusers, ATenFuseConv3dMul_Test) { - const auto graph_IR = gen_conv3d_mul_graph(); - auto fuser = std::make_shared(); - - fuse_test_helper(graph_IR, 3, fuser, {1, 2, 3, 4, 5}, {3, 2, 3, 3, 3}, {3}); - fuse_test_helper(graph_IR, 3, fuser, {3, 5, 8, 4, 6}, {12, 5, 3, 3, 3}, {12}); - fuse_test_helper(graph_IR, 3, fuser, {1, 2, 3, 4, 5}, {3, 2, 3, 3, 3}, {3}); -} - -TEST(Fusers, ATenFuseConv2dMul_Test) { - const auto graph_IR = gen_conv2d_mul_graph(); - auto fuser = std::make_shared(); - - fuse_test_helper(graph_IR, 2, fuser, {1, 2, 4, 5}, {3, 2, 3, 3}, {3}); - fuse_test_helper(graph_IR, 2, fuser, {3, 5, 4, 6}, {12, 5, 3, 3}, {12}); - fuse_test_helper(graph_IR, 2, fuser, {1, 2, 4, 5}, {3, 2, 3, 3}, {3}); -} - -TEST(Fusers, ATenFuseConv1dMul_Test) { - const auto graph_IR = gen_conv1d_mul_graph(); - auto fuser = std::make_shared(); - - fuse_test_helper(graph_IR, 1, fuser, {1, 2, 5}, {3, 2, 3}, {3}); - fuse_test_helper(graph_IR, 1, fuser, {3, 5, 6}, {12, 5, 3}, {12}); - fuse_test_helper(graph_IR, 1, fuser, {1, 2, 5}, {3, 2, 3}, {3}); -} \ No newline at end of file diff --git a/poros/unittest/op_fuser/fuse_copy_test.cpp b/poros/unittest/op_fuser/fuse_copy_test.cpp deleted file mode 100644 index 7a0ebdae82c..00000000000 --- a/poros/unittest/op_fuser/fuse_copy_test.cpp +++ /dev/null @@ -1,380 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file fuse_copy_test.cpp -* @author tianjinjin@baidu.com -* @date Mon Aug 22 10:47:14 CST 2022 -* @brief -**/ - -#include -#include - -#include "poros/lowering/fuse_copy.h" -#include "poros/lowering/op_fuse_pass.h" -#include "poros/util/graph_test_helper.h" - -static void fuse_test_helper(const std::string &graph_IR, - std::shared_ptr fuser, - std::vector input_shape, - bool with_single_value -) { - std::vector input_data; - input_data.push_back(at::randn(input_shape, {at::kCPU})); - std::vector input_data_type_mask = { - baidu::mirana::poros::graphtester::InputTensor, - }; - - if (with_single_value) { - input_data.push_back(at::randn({1}, {at::kCPU})); - input_data_type_mask.push_back(baidu::mirana::poros::graphtester::ConstantTensor); - } - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // poros_option.debug = true; - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector fused_output; - ASSERT_TRUE(baidu::mirana::poros::graphtester::run_graph_and_fused_graph(graph_IR, poros_option, fuser, - input_data, input_data_type_mask, - graph_output, fused_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, fused_output.size()); - ASSERT_TRUE(baidu::mirana::poros::graphtester::almost_equal(graph_output[0], fused_output[0], 1e-6)); -} - -/** - * this IR is generated from python code below: -def shift(x, n_segment, fold_div=3, inplace=False): - nt, c, h, w = x.size() - n_batch = nt // n_segment - x = x.view(n_batch, n_segment, c, h, w) - - fold = c // fold_div - - out = torch.zeros_like(x) - out[:, :-1, :fold] = x[:, 1:, :fold] - return out.view(nt, c, h, w) - * **/ -static std::string gen_simple_slice_graph() { - std::string graph = R"IR( - graph(%x : Tensor): - %none : NoneType = prim::Constant() - %0 : int = prim::Constant[value=0]() - %1 : int = prim::Constant[value=1]() - %2 : int = prim::Constant[value=2]() - %3 : int = prim::Constant[value=-1]() - %8 : int = prim::Constant[value=8]() - %16 : int = prim::Constant[value=16]() - %false : bool = prim::Constant[value=0]() - %fold : int = prim::Constant[value=21]() - - %292 : int[] = aten::size(%x) - %nt.3 : int, %c.3 : int, %h.3 : int, %w.3 : int = prim::ListUnpack(%292) - %n_batch.3 : int = aten::floordiv(%nt.3, %16) - %298 : int[] = prim::ListConstruct(%n_batch.3, %16, %c.3, %h.3, %w.3) - %x1.7 : Tensor = aten::view(%x, %298) - - %out : Tensor = aten::zeros_like(%x1.7, %none, %none, %none, %none, %none) # temporal_shift.py:12:18 - - %302 : Tensor = aten::slice(%x1.7, %0, %none, %none, %1) # temporal_shift.py:13:33 - %303 : Tensor = aten::slice(%302, %1, %1, %none, %1) # temporal_shift.py:13:33 - %304 : Tensor = aten::slice(%303, %2, %none, %fold, %1) # temporal_shift.py:13:33 - - %305 : Tensor = aten::slice(%out, %0, %none, %none, %1) # temporal_shift.py:13:12 - %306 : Tensor = aten::slice(%305, %1, %none, %3, %1) # temporal_shift.py:13:12 - %307 : Tensor = aten::slice(%306, %2, %none, %fold, %1) # temporal_shift.py:13:12 - %308 : Tensor = aten::copy_(%307, %304, %false) # temporal_shift.py:13:12 - - %322 : int[] = prim::ListConstruct(%nt.3, %c.3, %h.3, %w.3) - %x0.7 : Tensor = aten::view(%out, %322) - return (%x0.7))IR"; - return graph; -} - -/** - * this IR is generated from python code below: -def shift(x, n_segment, fold_div=3, inplace=False): - nt, c, h, w = x.size() - n_batch = nt // n_segment - x = x.view(n_batch, n_segment, c, h, w) - - fold = c // fold_div - - out = torch.zeros_like(x) - out[:, :-1, :fold] = x[:, 1:, :fold] - out[:, 1:, fold: 2 * fold] = x[:, :-1, fold: 2 * fold] - out[:, :, 2 * fold:] = x[:, :, 2 * fold:] - return out.view(nt, c, h, w) - * **/ -static std::string gen_complex_slice_graph() { - std::string graph = R"IR( - graph(%x : Tensor): - %none : NoneType = prim::Constant() - %0 : int = prim::Constant[value=0]() - %1 : int = prim::Constant[value=1]() - %2 : int = prim::Constant[value=2]() - %3 : int = prim::Constant[value=-1]() - %8 : int = prim::Constant[value=8]() - %16 : int = prim::Constant[value=16]() - %false : bool = prim::Constant[value=0]() - %fold : int = prim::Constant[value=21]() - - %292 : int[] = aten::size(%x) - %nt.3 : int, %c.3 : int, %h.3 : int, %w.3 : int = prim::ListUnpack(%292) - %n_batch.3 : int = aten::floordiv(%nt.3, %16) - %298 : int[] = prim::ListConstruct(%n_batch.3, %16, %c.3, %h.3, %w.3) - %x1.7 : Tensor = aten::view(%x, %298) - - %out : Tensor = aten::zeros_like(%x1.7, %none, %none, %none, %none, %none) # temporal_shift.py:12:18 - - %302 : Tensor = aten::slice(%x1.7, %0, %none, %none, %1) # temporal_shift.py:13:33 - %303 : Tensor = aten::slice(%302, %1, %1, %none, %1) # temporal_shift.py:13:33 - %304 : Tensor = aten::slice(%303, %2, %none, %fold, %1) # temporal_shift.py:13:33 - - %305 : Tensor = aten::slice(%out, %0, %none, %none, %1) # temporal_shift.py:13:12 - %306 : Tensor = aten::slice(%305, %1, %none, %3, %1) # temporal_shift.py:13:12 - %307 : Tensor = aten::slice(%306, %2, %none, %fold, %1) # temporal_shift.py:13:12 - %308 : Tensor = aten::copy_(%307, %304, %false) # temporal_shift.py:13:12 - - %309 : Tensor = aten::slice(%302, %1, %none, %3, %1) # temporal_shift.py:14:41 - %310 : int = aten::mul(%2, %fold) # temporal_shift.py:14:57 - %311 : Tensor = aten::slice(%309, %2, %fold, %310, %1) # temporal_shift.py:14:41 - - %312 : Tensor = aten::slice(%out, %0, %none, %none, %1) # temporal_shift.py:14:12 - %313 : Tensor = aten::slice(%312, %1, %1, %none, %1) # temporal_shift.py:14:12 - %314 : Tensor = aten::slice(%313, %2, %fold, %310, %1) # temporal_shift.py:14:12 - %315 : Tensor = aten::copy_(%314, %311, %false) # temporal_shift.py:14:12 - - %316 : Tensor = aten::slice(%302, %1, %none, %none, %1) # temporal_shift.py:15:35 - %317 : Tensor = aten::slice(%316, %2, %310, %none, %1) # temporal_shift.py:15:35 - - %318 : Tensor = aten::slice(%out, %0, %none, %none, %1) # temporal_shift.py:15:12 - %319 : Tensor = aten::slice(%318, %1, %none, %none, %1) # temporal_shift.py:15:12 - %320 : Tensor = aten::slice(%319, %2, %310, %none, %1) # temporal_shift.py:15:12 - %321 : Tensor = aten::copy_(%320, %317, %false) # temporal_shift.py:15:12 - - %322 : int[] = prim::ListConstruct(%nt.3, %c.3, %h.3, %w.3) - %x0.7 : Tensor = aten::view(%out, %322) - return (%x0.7))IR"; - return graph; -} - -/** - * this IR is generated from python code below: - * class SliceTest(torch.nn.Module): - def __init__(self): - super(SliceTest, self).__init__() - - def forward(self, x): - size = x.size() - #resize = size[:-1] - attention_mask = torch.zeros(size) - attention_mask[2:3:1, 2, :, 0, :] = 1 - out = attention_mask * 3 - return out - * **/ -static std::string gen_select_graph_with_single_value() { - std::string graph = R"IR( - graph(%x.1 : Tensor, %value : Tensor): - %33 : bool = prim::Constant[value=0]() - %5 : NoneType = prim::Constant() - %10 : int = prim::Constant[value=1]() # ../../test.py:11:44 - %12 : int = prim::Constant[value=2]() # ../../test.py:11:30 - %13 : int = prim::Constant[value=0]() # ../../test.py:11:36 - %15 : int = prim::Constant[value=3]() # ../../test.py:11:25 - %size.1 : int[] = aten::size(%x.1) # ../../test.py:8:15 - %attention_mask.1 : Tensor = aten::zeros(%size.1, %5, %5, %5, %5) # ../../test.py:10:25 - %16 : Tensor = aten::slice(%attention_mask.1, %13, %12, %15, %10) # ../../test.py:11:8 - %18 : Tensor = aten::select(%16, %10, %12) # ../../test.py:11:8 - %23 : Tensor = aten::slice(%18, %10, %5, %5, %10) # ../../test.py:11:8 - %25 : Tensor = aten::select(%23, %12, %13) # ../../test.py:11:8 - %30 : Tensor = aten::slice(%25, %12, %5, %5, %10) # ../../test.py:11:8 - %36 : Tensor = aten::copy_(%30, %value, %33) # ../../test.py:11:8 - %out.1 : Tensor = aten::mul(%attention_mask.1, %15) # ../../test.py:12:14 - return (%out.1))IR"; - return graph; -} - -/** - * this IR is generated from python code below: - * class SliceTest(torch.nn.Module): - def __init__(self): - super(SliceTest, self).__init__() - - def forward(self, x): - size = x.size() - #resize = size[:-1] - attention_mask = torch.zeros(size) - attention_mask[0, 2, 1:4:1, 0, :] = 1 - out = attention_mask * 3 - return out - * **/ -static std::string gen_select_graph_with_single_value2() { - std::string graph = R"IR( - graph(%x.1 : Tensor, %value : Tensor): - %30 : bool = prim::Constant[value=0]() - %5 : NoneType = prim::Constant() - %10 : int = prim::Constant[value=1]() # ../../test.py:11:44 - %12 : int = prim::Constant[value=0]() # ../../test.py:11:23 - %13 : int = prim::Constant[value=2]() # ../../test.py:11:26 - %19 : int = prim::Constant[value=4]() # ../../test.py:11:31 - %35 : int = prim::Constant[value=3]() # ../../test.py:12:31 - %size.1 : int[] = aten::size(%x.1) # ../../test.py:8:15 - %attention_mask.1 : Tensor = aten::zeros(%size.1, %5, %5, %5, %5) # ../../test.py:10:25 - %15 : Tensor = aten::select(%attention_mask.1, %12, %12) # ../../test.py:11:8 - %17 : Tensor = aten::select(%15, %12, %13) # ../../test.py:11:8 - %20 : Tensor = aten::slice(%17, %12, %10, %19, %10) # ../../test.py:11:8 - %22 : Tensor = aten::select(%20, %10, %12) # ../../test.py:11:8 - %27 : Tensor = aten::slice(%22, %10, %5, %5, %10) # ../../test.py:11:8 - %33 : Tensor = aten::copy_(%27, %value, %30) # ../../test.py:11:8 - %out.1 : Tensor = aten::mul(%attention_mask.1, %35) # ../../test.py:12:14 - return (%out.1))IR"; - return graph; -} - -/** - * this IR is generated from python code below: -class ClipBoxes(torch.nn.Module): - def __init__(self): - super(ClipBoxes, self).__init__() - - def forward(self, boxes): - boxes[:, :, 0] = torch.clamp(boxes[:, :, 0], min=0) - return boxes * 2 - * **/ -static std::string gen_select_with_tensor_value() { - std::string graph = R"IR( - graph(%boxes.1 : Tensor): - %31 : bool = prim::Constant[value=0]() - %6 : NoneType = prim::Constant() - %5 : int = prim::Constant[value=1]() # test.py:54:37 - %3 : int = prim::Constant[value=0]() # test.py:54:49 - %34 : int = prim::Constant[value=2]() # test.py:60:23 - %8 : Tensor = aten::slice(%boxes.1, %3, %6, %6, %5) # test.py:54:37 - %13 : Tensor = aten::slice(%8, %5, %6, %6, %5) # test.py:54:37 - %15 : Tensor = aten::select(%13, %34, %3) # test.py:54:37 - %17 : Tensor = aten::clamp(%15, %3, %6) # test.py:54:25 - %23 : Tensor = aten::slice(%boxes.1, %3, %6, %6, %5) # test.py:54:8 - %28 : Tensor = aten::slice(%23, %5, %6, %6, %5) # test.py:54:8 - %30 : Tensor = aten::select(%28, %34, %3) # test.py:54:8 - %32 : Tensor = aten::copy_(%30, %17, %31) # test.py:54:8 - %35 : Tensor = aten::mul(%boxes.1, %34) # test.py:60:15 - return (%35))IR"; - return graph; -} - -/** - * this IR is generated from python code below: -class ClipBoxes(torch.nn.Module): - def __init__(self): - super(ClipBoxes, self).__init__() - - def forward(self, boxes): - boxes[:, 0, :, :] = torch.clamp(boxes[:, 0, :, :], min=0) - return boxes * 2 - * **/ -static std::string gen_select_with_tensor_value2() { - std::string graph = R"IR( - graph(%boxes.1 : Tensor): - %41 : bool = prim::Constant[value=0]() - %6 : NoneType = prim::Constant() - %5 : int = prim::Constant[value=1]() # test.py:46:40 - %3 : int = prim::Constant[value=0]() # test.py:46:49 - %44 : int = prim::Constant[value=2]() # test.py:47:23 - %8 : Tensor = aten::slice(%boxes.1, %3, %6, %6, %5) # test.py:46:40 - %10 : Tensor = aten::select(%8, %5, %3) # test.py:46:40 - %15 : Tensor = aten::slice(%10, %5, %6, %6, %5) # test.py:46:40 - %20 : Tensor = aten::slice(%15, %44, %6, %6, %5) # test.py:46:40 - %22 : Tensor = aten::clamp(%20, %3, %6) # test.py:46:28 - %28 : Tensor = aten::slice(%boxes.1, %3, %6, %6, %5) # test.py:46:8 - %30 : Tensor = aten::select(%28, %5, %3) # test.py:46:8 - %35 : Tensor = aten::slice(%30, %5, %6, %6, %5) # test.py:46:8 - %40 : Tensor = aten::slice(%35, %44, %6, %6, %5) # test.py:46:8 - %42 : Tensor = aten::copy_(%40, %22, %41) # test.py:46:8 - %45 : Tensor = aten::mul(%boxes.1, %44) # test.py:47:15 - return (%45))IR"; - return graph; -} - -/** - * this IR is generated from python code below: -class ClipBoxes(torch.nn.Module): - def __init__(self): - super(ClipBoxes, self).__init__() - - def forward(self, boxes): - boxes[:, 0, :, 1] = torch.clamp(boxes[:, 0, :, 1], min=0) - return boxes * 2 - * **/ -static std::string gen_select_with_tensor_value3() { - std::string graph = R"IR( - graph(%boxes.1 : Tensor): - %36 : bool = prim::Constant[value=0]() - %7 : NoneType = prim::Constant() - %3 : int = prim::Constant[value=0]() # test.py:50:49 - %4 : int = prim::Constant[value=1]() # test.py:50:55 - %39 : int = prim::Constant[value=2]() # test.py:51:23 - %9 : Tensor = aten::slice(%boxes.1, %3, %7, %7, %4) # test.py:50:40 - %11 : Tensor = aten::select(%9, %4, %3) # test.py:50:40 - %16 : Tensor = aten::slice(%11, %4, %7, %7, %4) # test.py:50:40 - %18 : Tensor = aten::select(%16, %39, %4) # test.py:50:40 - %20 : Tensor = aten::clamp(%18, %3, %7) # test.py:50:28 - %26 : Tensor = aten::slice(%boxes.1, %3, %7, %7, %4) # test.py:50:8 - %28 : Tensor = aten::select(%26, %4, %3) # test.py:50:8 - %33 : Tensor = aten::slice(%28, %4, %7, %7, %4) # test.py:50:8 - %35 : Tensor = aten::select(%33, %39, %4) # test.py:50:8 - %37 : Tensor = aten::copy_(%35, %20, %36) # test.py:50:8 - %40 : Tensor = aten::mul(%boxes.1, %39) # test.py:51:15 - return (%40))IR"; - return graph; -} - -TEST(Fusers, ATenFuseCopySliceTest) { - auto fuser = std::make_shared(); - //situation1: out[:, :-1, :fold] = x[:, 1:, :fold] - const auto slice_graph_simple = gen_simple_slice_graph(); - fuse_test_helper(slice_graph_simple, fuser, {16, 64, 16, 16}, false); - //situation2: multi-copy - const auto slice_graph = gen_complex_slice_graph(); - fuse_test_helper(slice_graph, fuser, {16, 64, 16, 16}, false); -} - -TEST(Fusers, ATenFuseCopySelectWithSingleValueTest) { - auto fuser = std::make_shared(); - //situation1: attention_mask[2:3:1, 2, :, 0, :] = 1 - const auto select_graph_IR = gen_select_graph_with_single_value(); - fuse_test_helper(select_graph_IR, fuser, {4, 4, 5, 4, 3}, true); - - //situation2: attention_mask[0, 2, 1:4:1, 0, :] = 1 - const auto select_graph_IR2 = gen_select_graph_with_single_value2(); - fuse_test_helper(select_graph_IR2, fuser, {4, 4, 5, 4, 3}, true); -} - -TEST(Fusers, ATenFuseCopySelectWithTensorValueTest) { - auto fuser = std::make_shared(); - //situation1: boxes[:, :, 0] = torch.clamp(boxes[:, :, 0], min=0) - const auto select_graph_IR= gen_select_with_tensor_value(); - fuse_test_helper(select_graph_IR, fuser, {1, 20, 4}, false); - - //situation2: boxes[:, 0, :, :] = torch.clamp(boxes[:, 0, :, :], min=0) - const auto select_graph_IR2 = gen_select_with_tensor_value2(); - fuse_test_helper(select_graph_IR2, fuser, {1, 20, 4, 5}, false); - - //situation3: boxes[:, 0, :, 1] = torch.clamp(boxes[:, 0, :, 1], min=0) - const auto select_graph_IR3 = gen_select_with_tensor_value3(); - fuse_test_helper(select_graph_IR3, fuser, {1, 20, 4, 5}, false); -} \ No newline at end of file diff --git a/poros/unittest/op_fuser/fuse_hard_swish_test.cpp b/poros/unittest/op_fuser/fuse_hard_swish_test.cpp deleted file mode 100644 index 0719a033f6a..00000000000 --- a/poros/unittest/op_fuser/fuse_hard_swish_test.cpp +++ /dev/null @@ -1,71 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file fuse_hard_swish_test.cpp -* @author Lin Xiao Chun (linxiaochun@baidu.com) -* @date 2022-04-07 15:31:03 -* @brief -**/ - -#include -#include - -#include "poros/lowering/fuse_hard_swish.h" -#include "poros/lowering/op_fuse_pass.h" -#include "poros/util/graph_test_helper.h" - -static void fuse_test_helper(const std::string &graph_IR, - std::shared_ptr fuser, - std::vector input_shape -) { - std::vector input_data; - input_data.push_back(at::randn(input_shape, {at::kCPU})); - - const std::vector input_data_type_mask = { - baidu::mirana::poros::graphtester::InputTensor, - }; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector fused_output; - ASSERT_TRUE(baidu::mirana::poros::graphtester::run_graph_and_fused_graph(graph_IR, poros_option, fuser, - input_data, input_data_type_mask, - graph_output, fused_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, fused_output.size()); - ASSERT_TRUE(baidu::mirana::poros::graphtester::almost_equal(graph_output[0], fused_output[0], 1e-6)); -} - -static std::string gen_hardswish_graph() { - - std::string hardsiwsh = R"IR( - graph(%x): - %out: Tensor = aten::hardswish(%x) - return (%out))IR"; - return hardsiwsh; -} - -TEST(Fusers, ATenFuseHardSwish_Test) { - const auto graph_IR = gen_hardswish_graph(); - auto fuser = std::make_shared(); - - fuse_test_helper(graph_IR, fuser, {2, 3, 4, 5}); - fuse_test_helper(graph_IR, fuser, {3, 4, 5}); - fuse_test_helper(graph_IR, fuser, {4, 5}); - fuse_test_helper(graph_IR, fuser, {5}); -} - diff --git a/poros/unittest/op_fuser/fuse_meshgrid_test.cpp b/poros/unittest/op_fuser/fuse_meshgrid_test.cpp deleted file mode 100644 index 0ebc44538d7..00000000000 --- a/poros/unittest/op_fuser/fuse_meshgrid_test.cpp +++ /dev/null @@ -1,77 +0,0 @@ -// Copyright (c) 2022 Baidu, Inc. All Rights Reserved. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -/** -* @file fuse_meshgrid_test.cpp -* @author Lin Xiao Chun (linxiaochun@baidu.com) -* @date 2022-04-29 14:56:38 -* @brief -**/ - -#include -#include - -#include "poros/lowering/fuse_meshgrid.h" -#include "poros/lowering/op_fuse_pass.h" -#include "poros/util/graph_test_helper.h" - -static void fuse_test_helper(const std::string &graph_IR, - std::shared_ptr fuser, - std::pair input_shape -) { - std::vector input_data; - input_data.push_back(at::randn(input_shape.first, {at::kCPU})); - input_data.push_back(at::randn(input_shape.second, {at::kCPU})); - const std::vector input_data_type_mask = { - baidu::mirana::poros::graphtester::InputTensor, - baidu::mirana::poros::graphtester::InputTensor, - }; - - baidu::mirana::poros::PorosOptions poros_option; // default device GPU - // 运行原图与engine获取结果 - std::vector graph_output; - std::vector fused_output; - ASSERT_TRUE(baidu::mirana::poros::graphtester::run_graph_and_fused_graph(graph_IR, poros_option, fuser, - input_data, input_data_type_mask, - graph_output, fused_output)); - - ASSERT_EQ(1, graph_output.size()); - ASSERT_EQ(1, fused_output.size()); - ASSERT_TRUE(baidu::mirana::poros::graphtester::almost_equal(graph_output[0], fused_output[0], 1e-6)); -} - -static std::string gen_meshgrid_graph() { - - std::string graph = R"IR( - graph(%x.1 : Tensor, - %y.1 : Tensor): - %10 : int = prim::Constant[value=1]() - %4 : Tensor[] = prim::ListConstruct(%x.1, %y.1) - %5 : Tensor[] = aten::meshgrid(%4) - %grid_x.1 : Tensor, %grid_y.1 : Tensor = prim::ListUnpack(%5) - %11 : Tensor = aten::add(%grid_x.1, %grid_y.1, %10) - return (%11))IR"; - return graph; -} - -TEST(Fusers, ATenFuseMeshgrid_Test) { - const auto graph_IR = gen_meshgrid_graph(); - auto fuser = std::make_shared(); - - fuse_test_helper(graph_IR, fuser, {2, 3}); - fuse_test_helper(graph_IR, fuser, {100, 200}); - fuse_test_helper(graph_IR, fuser, {1000, 1000}); - fuse_test_helper(graph_IR, fuser, {1,2}); -} - diff --git a/scripts/build.py b/scripts/build.py new file mode 100755 index 00000000000..c91b6f48b9d --- /dev/null +++ b/scripts/build.py @@ -0,0 +1,273 @@ +#!/usr/bin/env python3 +# -*- coding: utf-8 -*- +""" +Cross-platform build and packaging script for FastDeploy (fastdeploy_ppocr). +Usage: + python scripts/build.py --target [macos-arm64 | linux-x64 | windows-x64 | windows-arm64 | android-arm64 | python-wheel] +""" + +import os +import sys +import shutil +import argparse +import subprocess +import urllib.request +import zipfile +import tarfile + +PROJECT_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "..")) +DEPS_DIR = os.path.join(PROJECT_ROOT, "deps") +BUILD_DIR = os.path.join(PROJECT_ROOT, "build") +DIST_DIR = os.path.join(PROJECT_ROOT, "dist") +ORT_VERSION = "1.29.0" + +def log(msg): + print(f"\n[build.py] ===> {msg}", flush=True) + +def run_cmd(cmd, cwd=PROJECT_ROOT, env=None): + log(f"Running command: {' '.join(cmd) if isinstance(cmd, list) else cmd}") + cmd_env = os.environ.copy() + if env: + cmd_env.update(env) + ret = subprocess.run(cmd, cwd=cwd, env=cmd_env, shell=isinstance(cmd, str)) + if ret.returncode != 0: + print(f"\n[build.py] Error: command failed with return code {ret.returncode}", file=sys.stderr) + sys.exit(ret.returncode) + +def download_file(url, target_path): + if os.path.exists(target_path): + log(f"File already exists: {target_path}") + return + log(f"Downloading {url} to {target_path} ...") + os.makedirs(os.path.dirname(target_path), exist_ok=True) + req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"}) + with urllib.request.urlopen(req) as resp, open(target_path, "wb") as out_file: + shutil.copyfileobj(resp, out_file) + log(f"Download complete: {target_path}") + +def extract_archive(archive_path, extract_to): + log(f"Extracting {archive_path} to {extract_to} ...") + os.makedirs(extract_to, exist_ok=True) + if archive_path.endswith(".zip") or archive_path.endswith(".nupkg") or archive_path.endswith(".aar"): + with zipfile.ZipFile(archive_path, 'r') as zip_ref: + zip_ref.extractall(extract_to) + elif archive_path.endswith(".tar.gz") or archive_path.endswith(".tgz"): + with tarfile.open(archive_path, 'r:gz') as tar_ref: + tar_ref.extractall(extract_to) + else: + # Fallback for 7z / exe on Windows + if shutil.which("7z"): + run_cmd(["7z", "x", archive_path, f"-o{extract_to}", "-y"]) + else: + raise RuntimeError(f"Unsupported archive or 7z not installed: {archive_path}") + +def package_dist(target_name, fmt="tar.gz"): + log(f"Packaging dist directory to fastdeploy_ppocr-{target_name}.{fmt} ...") + pkg_name = f"fastdeploy_ppocr-{target_name}" + if fmt == "tar.gz": + archive_file = os.path.join(PROJECT_ROOT, f"{pkg_name}.tar.gz") + with tarfile.open(archive_file, "w:gz") as tar: + for item in os.listdir(DIST_DIR): + item_path = os.path.join(DIST_DIR, item) + tar.add(item_path, arcname=item) + elif fmt == "zip": + archive_file = os.path.join(PROJECT_ROOT, f"{pkg_name}.zip") + shutil.make_archive(os.path.join(PROJECT_ROOT, pkg_name), 'zip', DIST_DIR) + log(f"Artifact created: {pkg_name}.{fmt}") + +def clean(): + for d in [BUILD_DIR, DIST_DIR]: + if os.path.exists(d): + log(f"Cleaning {d} ...") + shutil.rmtree(d, ignore_errors=True) + +# ------------------------------------------------------------------------- +# Targets +# ------------------------------------------------------------------------- + +def build_macos_arm64(args): + clean() + os.makedirs(BUILD_DIR, exist_ok=True) + # Query brew prefix + def get_brew_prefix(pkg): + res = subprocess.run(["brew", "--prefix", pkg], capture_output=True, text=True) + return res.stdout.strip() if res.returncode == 0 else f"/opt/homebrew/opt/{pkg}" + + # Prefer opencv@4 over opencv for API compatibility + opencv_prefix = None + for pkg in ["opencv@4", "opencv"]: + prefix = get_brew_prefix(pkg) + if os.path.exists(prefix): + opencv_prefix = prefix + break + if not opencv_prefix: + opencv_prefix = get_brew_prefix("opencv@4") + + ort_prefix = get_brew_prefix("onnxruntime") + eigen_prefix = get_brew_prefix("eigen") + + cmake_args = [ + "cmake", "..", "-G", "Ninja", + f"-DCMAKE_BUILD_TYPE={args.build_type}", + f"-DCMAKE_INSTALL_PREFIX={DIST_DIR}", + f"-DCMAKE_PREFIX_PATH={opencv_prefix};{ort_prefix};{eigen_prefix}", + "-DWITH_CUDA=OFF" + ] + opencv_cmake_dir = os.path.join(opencv_prefix, "lib", "cmake", "opencv4") + if os.path.exists(opencv_cmake_dir): + cmake_args.append(f"-DOpenCV_DIR={opencv_cmake_dir}") + run_cmd(cmake_args, cwd=BUILD_DIR) + run_cmd(["ninja"], cwd=BUILD_DIR) + run_cmd(["cmake", "--install", ".", "--prefix", DIST_DIR], cwd=BUILD_DIR) + package_dist("macos-arm64", fmt="tar.gz") + +def build_linux_x64(args): + clean() + os.makedirs(BUILD_DIR, exist_ok=True) + ort_tar = os.path.join(DEPS_DIR, f"onnxruntime-linux-x64-{ORT_VERSION}.tgz") + ort_dir = os.path.join(DEPS_DIR, f"onnxruntime-linux-x64-{ORT_VERSION}") + if not os.path.exists(ort_dir): + download_file(f"https://github.com/microsoft/onnxruntime/releases/download/v{ORT_VERSION}/onnxruntime-linux-x64-{ORT_VERSION}.tgz", ort_tar) + extract_archive(ort_tar, DEPS_DIR) + + cmake_args = [ + "cmake", "..", "-G", "Ninja", + f"-DCMAKE_BUILD_TYPE={args.build_type}", + f"-DCMAKE_INSTALL_PREFIX={DIST_DIR}", + f"-DCMAKE_PREFIX_PATH={ort_dir};/usr/include/eigen3", + "-DWITH_CUDA=OFF" + ] + run_cmd(cmake_args, cwd=BUILD_DIR) + run_cmd(["ninja"], cwd=BUILD_DIR) + run_cmd(["cmake", "--install", ".", "--prefix", DIST_DIR], cwd=BUILD_DIR) + package_dist("linux-x64", fmt="tar.gz") + +def build_windows_x64(args): + clean() + os.makedirs(BUILD_DIR, exist_ok=True) + # 1. Eigen3 + eigen_zip = os.path.join(DEPS_DIR, "eigen.zip") + eigen_dir = os.path.join(DEPS_DIR, "eigen-3.4.0") + if not os.path.exists(eigen_dir): + download_file("https://gitlab.com/libeigen/eigen/-/archive/3.4.0/eigen-3.4.0.zip", eigen_zip) + extract_archive(eigen_zip, DEPS_DIR) + + # 2. ONNXRuntime Windows x64 (official GitHub release) + ort_zip = os.path.join(DEPS_DIR, f"onnxruntime-win-x64-{ORT_VERSION}.zip") + ort_dir = os.path.join(DEPS_DIR, f"onnxruntime-win-x64-{ORT_VERSION}") + if not os.path.exists(ort_dir): + download_file(f"https://github.com/microsoft/onnxruntime/releases/download/v{ORT_VERSION}/onnxruntime-win-x64-{ORT_VERSION}.zip", ort_zip) + extract_archive(ort_zip, DEPS_DIR) + + # 3. OpenCV + opencv_exe = os.path.join(DEPS_DIR, "opencv.exe") + opencv_root = os.path.join(DEPS_DIR, "opencv", "build") + if not os.path.exists(opencv_root): + download_file("https://github.com/opencv/opencv/releases/download/4.9.0/opencv-4.9.0-windows.exe", opencv_exe) + extract_archive(opencv_exe, DEPS_DIR) + + opencv_vc16_lib = os.path.join(opencv_root, "x64", "vc16", "lib") + opencv_dir = opencv_vc16_lib if os.path.exists(opencv_vc16_lib) else opencv_root + + cmake_args = [ + "cmake", "..", "-G", "Ninja", + f"-DCMAKE_BUILD_TYPE={args.build_type}", + f"-DCMAKE_INSTALL_PREFIX={DIST_DIR}", + f"-DOpenCV_DIR={opencv_dir}", + "-DOpenCV_RUNTIME=vc16", + f"-DCMAKE_PREFIX_PATH={ort_dir};{eigen_dir}", + "-DWITH_CUDA=OFF" + ] + run_cmd(cmake_args, cwd=BUILD_DIR) + run_cmd(["ninja"], cwd=BUILD_DIR) + run_cmd(["cmake", "--install", ".", "--prefix", DIST_DIR, "--config", args.build_type], cwd=BUILD_DIR) + + # Copy runtime DLLs to output bin directory + dist_bin = os.path.join(DIST_DIR, "bin") + os.makedirs(dist_bin, exist_ok=True) + ort_lib_dir = os.path.join(ort_dir, "lib") + if os.path.exists(ort_lib_dir): + for f in os.listdir(ort_lib_dir): + if f.endswith(".dll"): + shutil.copy2(os.path.join(ort_lib_dir, f), dist_bin) + opencv_bin = os.path.join(opencv_root, "x64", "vc16", "bin") + if os.path.exists(opencv_bin): + for f in os.listdir(opencv_bin): + if f.endswith(".dll"): + shutil.copy2(os.path.join(opencv_bin, f), dist_bin) + + package_dist("windows-x64", fmt="zip") + +def build_android_arm64(args): + clean() + os.makedirs(BUILD_DIR, exist_ok=True) + ndk_home = os.environ.get("ANDROID_NDK_LATEST_HOME") or os.environ.get("ANDROID_NDK_HOME") or os.environ.get("ANDROID_NDK_ROOT") + if not ndk_home or not os.path.exists(ndk_home): + raise RuntimeError(f"Android NDK not found in environment (checked ANDROID_NDK_LATEST_HOME, ANDROID_NDK_HOME, ANDROID_NDK_ROOT)") + + # 1. Eigen3 + eigen_zip = os.path.join(DEPS_DIR, "eigen.zip") + eigen_dir = os.path.join(DEPS_DIR, "eigen-3.4.0") + if not os.path.exists(eigen_dir): + download_file("https://gitlab.com/libeigen/eigen/-/archive/3.4.0/eigen-3.4.0.zip", eigen_zip) + extract_archive(eigen_zip, DEPS_DIR) + + # 2. OpenCV Android + opencv_zip = os.path.join(DEPS_DIR, "opencv-android.zip") + opencv_dir = os.path.join(DEPS_DIR, "OpenCV-android-sdk", "sdk", "native", "jni") + if not os.path.exists(opencv_dir): + download_file("https://github.com/opencv/opencv/releases/download/4.9.0/opencv-4.9.0-android-sdk.zip", opencv_zip) + extract_archive(opencv_zip, DEPS_DIR) + + # 3. ORT Android + ort_aar = os.path.join(DEPS_DIR, f"ort-android-{ORT_VERSION}.aar") + ort_extract_dir = os.path.join(DEPS_DIR, f"ort-android-{ORT_VERSION}") + ort_include_dir = os.path.join(ort_extract_dir, "headers") + ort_lib_file = os.path.join(ort_extract_dir, "jni", "arm64-v8a", "libonnxruntime.so") + + if not os.path.exists(ort_lib_file): + download_file(f"https://repo1.maven.org/maven2/com/microsoft/onnxruntime/onnxruntime-android/{ORT_VERSION}/onnxruntime-android-{ORT_VERSION}.aar", ort_aar) + extract_archive(ort_aar, ort_extract_dir) + + toolchain = os.path.join(ndk_home, "build", "cmake", "android.toolchain.cmake") + cmake_args = [ + "cmake", "..", "-G", "Ninja", + f"-DCMAKE_TOOLCHAIN_FILE={toolchain}", + "-DANDROID_ABI=arm64-v8a", + "-DANDROID_PLATFORM=android-24", + "-DANDROID_STL=c++_shared", + f"-DCMAKE_BUILD_TYPE={args.build_type}", + f"-DCMAKE_INSTALL_PREFIX={DIST_DIR}", + f"-DOpenCV_DIR={opencv_dir}", + f"-Donnxruntime_INCLUDE_DIR={ort_include_dir}", + f"-Donnxruntime_LIBRARY={ort_lib_file}", + f"-DCMAKE_PREFIX_PATH={eigen_dir}", + "-DCMAKE_FIND_ROOT_PATH_MODE_PACKAGE=BOTH", + "-DCMAKE_FIND_ROOT_PATH_MODE_INCLUDE=BOTH", + "-DCMAKE_FIND_ROOT_PATH_MODE_LIBRARY=BOTH", + "-DWITH_CUDA=OFF" + ] + run_cmd(cmake_args, cwd=BUILD_DIR) + run_cmd(["ninja"], cwd=BUILD_DIR) + run_cmd(["cmake", "--install", ".", "--prefix", DIST_DIR], cwd=BUILD_DIR) + package_dist("android-arm64", fmt="tar.gz") + +def main(): + parser = argparse.ArgumentParser(description="FastDeploy Build & Package Helper") + parser.add_argument("--target", required=True, + choices=["macos-arm64", "linux-x64", "windows-x64", "android-arm64"], + help="Target platform to build for") + parser.add_argument("--build-type", default="Release", help="CMake build type (Release, Debug, etc.)") + args = parser.parse_args() + + targets = { + "macos-arm64": build_macos_arm64, + "linux-x64": build_linux_x64, + "windows-x64": build_windows_x64, + "android-arm64": build_android_arm64, + } + targets[args.target](args) + +if __name__ == "__main__": + main() + diff --git a/commit-prepare.sh b/scripts/commit-prepare.sh similarity index 100% rename from commit-prepare.sh rename to scripts/commit-prepare.sh diff --git a/.clang_format.hook b/scripts/hooks/.clang_format.hook similarity index 100% rename from .clang_format.hook rename to scripts/hooks/.clang_format.hook diff --git a/.cpplint_pre_commit.hook b/scripts/hooks/.cpplint_pre_commit.hook similarity index 100% rename from .cpplint_pre_commit.hook rename to scripts/hooks/.cpplint_pre_commit.hook diff --git a/llm/.dockerignore b/services/llm/.dockerignore similarity index 100% rename from llm/.dockerignore rename to services/llm/.dockerignore diff --git a/llm/README.md b/services/llm/README.md similarity index 100% rename from llm/README.md rename to services/llm/README.md diff --git a/llm/client/README.md b/services/llm/client/README.md similarity index 100% rename from llm/client/README.md rename to services/llm/client/README.md diff --git a/llm/client/fastdeploy_client/__init__.py b/services/llm/client/fastdeploy_client/__init__.py similarity index 100% rename from llm/client/fastdeploy_client/__init__.py rename to services/llm/client/fastdeploy_client/__init__.py diff --git a/llm/client/fastdeploy_client/chatbot.py b/services/llm/client/fastdeploy_client/chatbot.py similarity index 100% rename from llm/client/fastdeploy_client/chatbot.py rename to services/llm/client/fastdeploy_client/chatbot.py diff --git a/llm/client/fastdeploy_client/command.py b/services/llm/client/fastdeploy_client/command.py similarity index 100% rename from llm/client/fastdeploy_client/command.py rename to services/llm/client/fastdeploy_client/command.py diff --git a/llm/client/fastdeploy_client/message.py b/services/llm/client/fastdeploy_client/message.py similarity index 100% rename from llm/client/fastdeploy_client/message.py rename to services/llm/client/fastdeploy_client/message.py diff --git a/llm/client/fastdeploy_client/utils.py b/services/llm/client/fastdeploy_client/utils.py similarity index 100% rename from llm/client/fastdeploy_client/utils.py rename to services/llm/client/fastdeploy_client/utils.py diff --git a/llm/client/requirements.txt b/services/llm/client/requirements.txt similarity index 100% rename from llm/client/requirements.txt rename to services/llm/client/requirements.txt diff --git a/llm/client/setup.py b/services/llm/client/setup.py similarity index 100% rename from llm/client/setup.py rename to services/llm/client/setup.py diff --git a/llm/dockerfiles/Dockerfile_serving_cuda118_cudnn8 b/services/llm/dockerfiles/Dockerfile_serving_cuda118_cudnn8 similarity index 100% rename from llm/dockerfiles/Dockerfile_serving_cuda118_cudnn8 rename to services/llm/dockerfiles/Dockerfile_serving_cuda118_cudnn8 diff --git a/llm/dockerfiles/Dockerfile_serving_cuda123_cudnn9 b/services/llm/dockerfiles/Dockerfile_serving_cuda123_cudnn9 similarity index 100% rename from llm/dockerfiles/Dockerfile_serving_cuda123_cudnn9 rename to services/llm/dockerfiles/Dockerfile_serving_cuda123_cudnn9 diff --git a/llm/docs/FastDeploy_usage_tutorial.md b/services/llm/docs/FastDeploy_usage_tutorial.md similarity index 100% rename from llm/docs/FastDeploy_usage_tutorial.md rename to services/llm/docs/FastDeploy_usage_tutorial.md diff --git a/llm/requirements-dev.txt b/services/llm/requirements-dev.txt similarity index 100% rename from llm/requirements-dev.txt rename to services/llm/requirements-dev.txt diff --git a/llm/server/config/config.pbtxt b/services/llm/server/config/config.pbtxt similarity index 100% rename from llm/server/config/config.pbtxt rename to services/llm/server/config/config.pbtxt diff --git a/llm/server/requirements.txt b/services/llm/server/requirements.txt similarity index 100% rename from llm/server/requirements.txt rename to services/llm/server/requirements.txt diff --git a/llm/server/scripts/start_server.sh b/services/llm/server/scripts/start_server.sh similarity index 100% rename from llm/server/scripts/start_server.sh rename to services/llm/server/scripts/start_server.sh diff --git a/llm/server/scripts/stop_server.sh b/services/llm/server/scripts/stop_server.sh similarity index 100% rename from llm/server/scripts/stop_server.sh rename to services/llm/server/scripts/stop_server.sh diff --git a/llm/server/server/__init__.py b/services/llm/server/server/__init__.py similarity index 100% rename from llm/server/server/__init__.py rename to services/llm/server/server/__init__.py diff --git a/llm/server/server/checker.py b/services/llm/server/server/checker.py similarity index 100% rename from llm/server/server/checker.py rename to services/llm/server/server/checker.py diff --git a/llm/server/server/data/__init__.py b/services/llm/server/server/data/__init__.py similarity index 100% rename from llm/server/server/data/__init__.py rename to services/llm/server/server/data/__init__.py diff --git a/llm/server/server/data/processor.py b/services/llm/server/server/data/processor.py similarity index 82% rename from llm/server/server/data/processor.py rename to services/llm/server/server/data/processor.py index 423fe6b6140..175f33b2db1 100644 --- a/llm/server/server/data/processor.py +++ b/services/llm/server/server/data/processor.py @@ -143,6 +143,9 @@ def process_request(self, request, max_seq_len=None): request["eos_token_ids"] = [] request["eos_token_ids"].extend(get_eos_token_id(self.tokenizer, self.config.generation_config)) + if "stop_seqs" not in request or (isinstance(request["stop_seqs"], (list, tuple)) and len(request["stop_seqs"]) == 0): + self.update_stop_seq(request) + if "input_ids" not in request or \ (isinstance(request["input_ids"], (list, tuple)) and len(request["input_ids"]) == 0): if "text" in request: @@ -282,7 +285,7 @@ def _load_tokenizer(self): """ if self.config.use_hf_tokenizer: from transformers import AutoTokenizer - return AutoTokenizer.from_pretrained(self.config.model_dir, use_fast=False, vocab_file=os.path.join(self.config.model_dir, "sentencepiece.bpe.model")) + return AutoTokenizer.from_pretrained(self.config.model_dir, use_fast=False) else: from paddlenlp.transformers import AutoTokenizer return AutoTokenizer.from_pretrained(self.config.model_dir) @@ -334,3 +337,53 @@ def get_pad_id(self): if isinstance(self.tokenizer, (LlamaTokenizer, Llama3Tokenizer)) and not self.tokenizer.pad_token_id: return self.tokenizer.eos_token return self.tokenizer.pad_token_id + + def pad_batch_data(self, insts, pad_id=0, return_seq_len=False, return_array=True, pad_style="right"): + """Pad the instances to the max sequence length in batch.""" + if len(insts) == 0: + padded_insts = np.array([[]], dtype=np.int64) if return_array else [[]] + if return_seq_len: + seq_len = np.array([], dtype=np.int64) if return_array else [] + return padded_insts, seq_len + return padded_insts + + max_len = max(map(len, insts)) + if pad_style == "left": + padded_insts = [[pad_id] * (max_len - len(inst)) + list(inst) for inst in insts] + else: + padded_insts = [list(inst) + [pad_id] * (max_len - len(inst)) for inst in insts] + if return_array: + padded_insts = np.array(padded_insts, dtype=np.int64).reshape([-1, max_len]) + + if return_seq_len: + seq_len = [len(inst) for inst in insts] + if return_array: + seq_len = np.array(seq_len, dtype=np.int64).reshape(-1, 1) + return padded_insts, seq_len + return padded_insts + + def update_stop_seq(self, request): + """ + Update stop sequences from request. + """ + max_stop_seqs_num = int(os.getenv("MAX_STOP_SEQS_NUM", 5)) + stop_seqs_max_len = int(os.getenv("STOP_SEQS_MAX_LEN", 8)) + stop_seqs = [] + raw_stop_sequences = request.get("stop_sequences", []) or [] + for seq in raw_stop_sequences[:max_stop_seqs_num]: + if seq != self.tokenizer.eos_token_id: + token_ids = self.tokenizer.convert_tokens_to_ids(self.tokenizer.tokenize(seq)) + if len(token_ids) > stop_seqs_max_len: + token_ids = token_ids[:stop_seqs_max_len] + stop_seqs.append(token_ids) + if len(stop_seqs) == 0: + request["stop_seqs"] = [] + request["stop_seqs_len"] = [] + else: + request["stop_seqs"], request["stop_seqs_len"] = self.pad_batch_data( + stop_seqs, + pad_id=-1, + return_seq_len=True, + return_array=False + ) + data_processor_logger.debug(f"processed request: {request['stop_seqs'], request['stop_seqs_len']}") diff --git a/llm/server/server/engine/__init__.py b/services/llm/server/server/engine/__init__.py similarity index 100% rename from llm/server/server/engine/__init__.py rename to services/llm/server/server/engine/__init__.py diff --git a/llm/server/server/engine/config.py b/services/llm/server/server/engine/config.py similarity index 90% rename from llm/server/server/engine/config.py rename to services/llm/server/server/engine/config.py index 6f0e1964e21..3b9a88f0c94 100644 --- a/llm/server/server/engine/config.py +++ b/services/llm/server/server/engine/config.py @@ -19,6 +19,7 @@ from paddlenlp.generation import GenerationConfig from server.utils import model_server_logger +from dataclasses import dataclass class Config: @@ -203,6 +204,27 @@ def get_model_config(self): model_config_json = json.load(open(self.model_config_path, 'r', encoding='utf-8')) return model_config_json + def get_speculate_config(self): + """ + get speculate_decoding related config + + Returns: + SpeculateConfig: the speculate related config + """ + speculate_config = SpeculateConfig() + model_cfg = self.get_model_config() + if model_cfg.get("speculate_method", "None") != "None": + speculate_config.speculate_method = str(model_cfg["speculate_method"]) + speculate_config.speculate_max_draft_token_num = model_cfg[ + "speculate_max_draft_token_num"] + speculate_config.speculate_max_ngram_size = model_cfg[ + "speculate_max_ngram_size"] + + if speculate_config.speculate_method not in ["None", "inference_with_reference"]: + model_server_logger.error(f"Unsupport speculate method: {speculate_config.speculate_method}") + + return speculate_config + def read_from_config(self): """ reset model config from json file @@ -234,3 +256,10 @@ def get_unique_name(self, name): def __str__(self) -> str: return json.dumps(self.__dict__, indent=4) + + +@dataclass +class SpeculateConfig: + speculate_method: str = "None" + speculate_max_draft_token_num: int = 1 + speculate_max_ngram_size: int = 1 \ No newline at end of file diff --git a/llm/server/server/engine/engine.py b/services/llm/server/server/engine/engine.py similarity index 100% rename from llm/server/server/engine/engine.py rename to services/llm/server/server/engine/engine.py diff --git a/llm/server/server/engine/infer.py b/services/llm/server/server/engine/infer.py similarity index 82% rename from llm/server/server/engine/infer.py rename to services/llm/server/server/engine/infer.py index 66377f9468e..dd889cf5418 100644 --- a/llm/server/server/engine/infer.py +++ b/services/llm/server/server/engine/infer.py @@ -29,6 +29,7 @@ from paddlenlp_ops import step_paddle from server.data.processor import DataProcessor from server.engine.config import Config +from paddlenlp.experimental.transformers import InferenceWithReferenceProposer from server.utils import get_logger from task_queue_manager import TaskQueueManager @@ -46,12 +47,19 @@ def __init__(self, args): self.config = Config() self.model_cfg = self.config.get_model_config() + self.speculate_config = self.config.get_speculate_config() + self.is_speculate_decoding = self.speculate_config.speculate_method != "None" self.format_print_configuration() self.args.num_layers = self.get_value(self.model_cfg, ["num_hidden_layers", "num_layers"]) self.args.num_attention_heads = self.get_value(self.model_cfg, ["num_attention_heads", "n_head"]) self.args.hidden_size = self.model_cfg["hidden_size"] + self.reduce_dialogue_repetition = int(os.environ.get("REDUCE_DIALOGUE_REPETITION", 0)) + + self.max_stop_seqs_num = int(os.getenv("MAX_STOP_SEQS_NUM", 5)) + self.stop_seqs_max_len = int(os.getenv("STOP_SEQS_MAX_LEN", 8)) + self.nranks = dist.get_world_size() self.init_dist_env() self.rank = fleet.worker_index() @@ -62,6 +70,21 @@ def __init__(self, args): self.cache_kvs = {} self.init_inputs() + self.proposer = None + if self.is_speculate_decoding: + logger.info(f'Using speculate decoding, method: {self.speculate_config.speculate_method}.') + if self.speculate_config.speculate_method == "inference_with_reference": + self.proposer = InferenceWithReferenceProposer( + self.speculate_config.speculate_max_draft_token_num, + self.speculate_config.speculate_max_ngram_size, + self.args.max_batch_size, + self.args.max_seq_len) + else: + logger.error( + f'Unsupported speculate method: {self.speculate_config.speculate_method}, disabling speculate decoding.' + ) + self.is_speculate_decoding = False + self.infer_queue = TaskQueueManager(rank=self.rank, mp_num=self.nranks, port=self.config.infer_port) model_rank_path = os.path.join(self.args.model_dir, f"rank_{self.rank}") @@ -246,6 +269,33 @@ def init_inputs(self): self.share_inputs['free_list_len'] = paddle.full( shape=[1], fill_value=self.free_list_len, dtype="int32") + self.share_inputs['stop_seqs_len'] = paddle.full( + shape=[self.args.max_batch_size, self.max_stop_seqs_num], + fill_value=0, + dtype="int32") + self.share_inputs['stop_seqs'] = paddle.full( + shape=[self.args.max_batch_size, self.max_stop_seqs_num, self.stop_seqs_max_len], + fill_value=-1, + dtype="int64") + + if self.reduce_dialogue_repetition: + self.share_inputs["first_token_ids"] = paddle.full( + shape=[self.args.max_batch_size, 1], fill_value=-1, dtype="int64") + self.share_inputs["ori_seq_lens_encoder"] = paddle.full( + shape=[self.args.max_batch_size, 1], fill_value=0, dtype="int32") + # speculate decoding input + if self.is_speculate_decoding: + self.share_inputs["accept_tokens"] = paddle.full( + shape=[self.args.max_batch_size, self.speculate_config.speculate_max_draft_token_num + 1], fill_value=0, dtype="int64" + ) + self.share_inputs["accept_num"] = paddle.full(shape=[self.args.max_batch_size], fill_value=0, dtype="int32") + self.share_inputs["draft_tokens"] = paddle.full( + shape=[self.args.max_batch_size, self.speculate_config.speculate_max_draft_token_num + 1], fill_value=0, dtype="int64" + ) + self.share_inputs["actual_draft_token_num"] = paddle.full( + shape=[self.args.max_batch_size], fill_value=self.speculate_config.speculate_max_draft_token_num, dtype="int32" + ) + def dy_input_preprocess(self, tasks): """ dynamic insertion @@ -279,6 +329,10 @@ def dy_input_preprocess(self, tasks): self.share_inputs['max_length'][idx:idx + 1] = max_dec_len self.share_inputs['stop_flags'][idx:idx + 1] = False + if self.reduce_dialogue_repetition: + self.share_inputs['first_token_ids'][idx:idx + 1] = self.share_inputs['input_ids'][idx:idx + 1, :1] + self.share_inputs["ori_seq_lens_encoder"][idx:idx + 1] = length + if "infer_seed" in task: self.share_inputs['infer_seed'][idx:idx + 1] = task['infer_seed'] @@ -288,10 +342,49 @@ def dy_input_preprocess(self, tasks): self.share_inputs["block_tables"][idx:idx + 1, :encoder_block_num] = np.array( task['block_tables'], dtype="int32") + self.share_inputs['stop_seqs_len'][idx:idx + 1, :] = 0 + self.share_inputs['stop_seqs'][idx:idx + 1, :, :] = -1 + + if "stop_seqs_len" in task and "stop_seqs" in task: + raw_stop_seqs = task["stop_seqs"][:self.max_stop_seqs_num] + raw_stop_seqs_len = task["stop_seqs_len"][:self.max_stop_seqs_num] + stop_seqs_num = len(raw_stop_seqs_len) + if stop_seqs_num > 0 and len(raw_stop_seqs) > 0: + truncated_seqs = [] + truncated_lens = [] + for s_len, s_seq in zip(raw_stop_seqs_len, raw_stop_seqs): + seq_trunc = s_seq[:self.stop_seqs_max_len] + truncated_seqs.append(seq_trunc) + truncated_lens.append(min(s_len, self.stop_seqs_max_len)) + + padded_lens = list(truncated_lens) + for _ in range(len(padded_lens), self.max_stop_seqs_num): + padded_lens.append(0) + self.share_inputs['stop_seqs_len'][idx:idx + 1, :] = np.array( + padded_lens, dtype="int32") + + max_seq_len = max(len(s) for s in truncated_seqs) if truncated_seqs else 0 + if max_seq_len > 0: + padded_seqs = [ + list(s) + [-1] * (max_seq_len - len(s)) for s in truncated_seqs + ] + self.share_inputs['stop_seqs'][idx:idx + 1, :stop_seqs_num, :max_seq_len] = np.array( + padded_seqs, dtype="int64") + + if self.is_speculate_decoding: + self.share_inputs["draft_tokens"][idx:idx + 1] = np.zeros([self.speculate_config.speculate_max_draft_token_num + 1]) + self.share_inputs["actual_draft_token_num"][idx:idx + 1] = np.array([self.speculate_config.speculate_max_draft_token_num]) + def step_cuda(self, seq_lens_this_time): """ step cuda """ + # whether speculate decoding + if self.is_speculate_decoding: + speculate_step_token_num = self.speculate_config.speculate_max_draft_token_num + 1 + else: + speculate_step_token_num = 0 + step_paddle(self.share_inputs['stop_flags'], seq_lens_this_time, self.share_inputs['step_seq_lens_encoder'], self.share_inputs['seq_lens_encoder'], @@ -304,7 +397,8 @@ def step_cuda(self, seq_lens_this_time): self.share_inputs['free_list'], self.share_inputs['free_list_len'], self.share_inputs['input_ids'], self.share_inputs['pre_ids'], self.share_inputs['step_idx'], self.share_inputs['next_tokens'], - self.args.block_size, self.args.enc_dec_block_num, self.args.first_token_id) + self.args.block_size, self.args.enc_dec_block_num, self.args.first_token_id, + speculate_step_token_num) def initialize_engine_ready_check_flag(self): """ @@ -429,6 +523,13 @@ def run(self): time.sleep(0.001) continue + if self.proposer is not None: + self.proposer.run( + self.share_inputs, + real_batch_size=seq_lens_this_time.shape[0], + seq_lens_this_time=seq_lens_this_time, + ) + self.infer_engine.predictor.run() self.share_inputs['infer_seed'].add_(infer_seed_increment) self.share_inputs['infer_seed'][:] %= self.MAX_INFER_SEED @@ -474,6 +575,11 @@ def _init_predictor(self): config.switch_ir_optim(False) config.enable_use_cuda(100, device_id) + pir_flag = int(os.environ.get("FLAGS_enable_pir_api", 0)) + if pir_flag == 1: + config.enable_new_executor() + config.enable_new_ir() + # distributed config if self.mp_degree > 1: trainer_endpoints = fleet.worker_endpoints() diff --git a/llm/server/server/engine/resource_manager.py b/services/llm/server/server/engine/resource_manager.py similarity index 100% rename from llm/server/server/engine/resource_manager.py rename to services/llm/server/server/engine/resource_manager.py diff --git a/llm/server/server/engine/task_queue_manager.py b/services/llm/server/server/engine/task_queue_manager.py similarity index 100% rename from llm/server/server/engine/task_queue_manager.py rename to services/llm/server/server/engine/task_queue_manager.py diff --git a/llm/server/server/engine/token_processor.py b/services/llm/server/server/engine/token_processor.py similarity index 65% rename from llm/server/server/engine/token_processor.py rename to services/llm/server/server/engine/token_processor.py index 507a3d43bdf..396145429aa 100644 --- a/llm/server/server/engine/token_processor.py +++ b/services/llm/server/server/engine/token_processor.py @@ -20,8 +20,9 @@ from datetime import datetime import numpy as np -from paddlenlp_ops import get_output +from paddlenlp_ops import get_output, speculate_get_output from server.utils import datetime_diff, model_server_logger, monitor_logger +from paddlenlp.utils.env import MAX_DRAFT_TOKENS, SPECULATE_MAX_BSZ class TokenProcessor(object): @@ -37,7 +38,12 @@ def __init__(self, cfg): self.all_tokens = [[] for _ in range(self.cfg.max_batch_size)] self.tokens_counter = Counter() - self.output_tokens = paddle.full(shape=[self.cfg.max_batch_size + 2, 1], fill_value=2, dtype="int64") + + self.is_speculate_decoding = self.cfg.get_speculate_config().speculate_method != "None" + if self.is_speculate_decoding: + self.output_tokens = paddle.full(shape=[SPECULATE_MAX_BSZ * MAX_DRAFT_TOKENS + SPECULATE_MAX_BSZ + 2, 1], fill_value=2, dtype="int64") + else: + self.output_tokens = paddle.full(shape=[self.cfg.max_batch_size + 2, 1], fill_value=2, dtype="int64") self.worker = None self.record_time_interval = int(os.getenv("RECORD_TIME_INTERVAL", "600")) @@ -77,10 +83,14 @@ def process_sampling_results(self): try: rank_id = 0 is_blocking = True - get_output(self.output_tokens, rank_id, is_blocking) + if self.is_speculate_decoding: + speculate_get_output(self.output_tokens, rank_id, is_blocking) + else: + get_output(self.output_tokens, rank_id, is_blocking) if self.output_tokens[0, 0] == -2: continue + self._process_batch_output() except Exception as e: model_server_logger.info("while get input_data error: {0} {1}".format(e, str(traceback.format_exc()))) @@ -101,14 +111,14 @@ def postprocess(self, batch_result, exist_finished_task=False): with open(result_file, "a") as f: f.write("{}\n".format(result)) - def _get_single_result(self, i, task_id, token_id, task): + def _get_single_result(self, i, task_id, token_ids, task): """ processing single results Args: i (int): batch index task_id (str): task id - token_id (int): token id + token_ids (list): token id task (dict): task information Returns: @@ -121,7 +131,7 @@ def _get_single_result(self, i, task_id, token_id, task): result = { "req_id": task_id, "is_end": 0, - "token_ids": [token_id], + "token_ids": token_ids, "send_idx": self.tokens_counter[task_id], "inference_time_cost": inference_time_cost, "infer_seed": task["infer_seed"], @@ -137,26 +147,30 @@ def _get_single_result(self, i, task_id, token_id, task): result[key] = str(task[key]) # fill some extra information - if token_id in task["eos_token_ids"]: - result["is_end"] = 1 - result["token_ids"] = [] - result["tokens_all_num"] = len(self.all_tokens[i]) + 1 - result["tokens_all_ids"] = self.all_tokens[i] - - info_dict = {} - info_dict["req_id"] = task["req_id"] - info_dict["input_token_num"] = len(task["input_ids"]) - info_dict["output_token_num"] = len(self.all_tokens[i]) - if hasattr(task, "preprocess_start_time") and hasattr(task, "preprocess_end_time"): - info_dict["preprocess_cost_time"] = datetime_diff(task["preprocess_start_time"], - task["preprocess_end_time"]) - if hasattr(task, "preprocess_end_time") and hasattr(task, "schedule_start_time"): - info_dict["cache_waiting_cost_time"] = datetime_diff(task["preprocess_end_time"], - task["schedule_start_time"]) - info_dict["inference_time_cost"] = task["inference_time_cost"] - info_dict["version"] = "4.6" - info_dict["timestamp"] = time.time() - monitor_logger.info(f"{info_dict}") + result["token_ids"] = [] + for token_id in token_ids: + if token_id in task["eos_token_ids"]: + result["is_end"] = 1 + result["tokens_all_num"] = len(self.all_tokens[i]) + len(result["token_ids"]) + 1 + result["tokens_all_ids"] = self.all_tokens[i] + result["token_ids"] + + info_dict = {} + info_dict["req_id"] = task["req_id"] + info_dict["input_token_num"] = len(task["input_ids"]) + info_dict["output_token_num"] = len(self.all_tokens[i]) + len(result["token_ids"]) + if hasattr(task, "preprocess_start_time") and hasattr(task, "preprocess_end_time"): + info_dict["preprocess_cost_time"] = datetime_diff(task["preprocess_start_time"], + task["preprocess_end_time"]) + if hasattr(task, "preprocess_end_time") and hasattr(task, "schedule_start_time"): + info_dict["cache_waiting_cost_time"] = datetime_diff(task["preprocess_end_time"], + task["schedule_start_time"]) + info_dict["inference_time_cost"] = task["inference_time_cost"] + info_dict["version"] = "OpenSource" + info_dict["timestamp"] = time.time() + monitor_logger.info(f"{info_dict}") + break + else: + result["token_ids"].append(token_id) return result @@ -177,7 +191,10 @@ def _process_batch_output(self): """ tokens = self.output_tokens.numpy() batch = self.output_tokens[1, 0] - tokens = tokens[2:batch + 2] + if not self.is_speculate_decoding: + tokens = tokens[2:batch + 2] + else: + accept_num = tokens[2:batch + 2] batch_result = list() exist_finished_task = False @@ -185,25 +202,31 @@ def _process_batch_output(self): if self.resource_manager.stop_flags[i]: continue - token_id = int(tokens[i, 0]) - if token_id < 0: + if not self.is_speculate_decoding: + token_ids = [int(tokens[i, 0])] + else: + token_ids = tokens[2 + SPECULATE_MAX_BSZ + i * MAX_DRAFT_TOKENS: 2 + SPECULATE_MAX_BSZ + i * MAX_DRAFT_TOKENS + accept_num[i, 0], 0].tolist() + + if any(token_id < 0 for token_id in token_ids): continue task = self.resource_manager.tasks_list[i] task_id = task["req_id"] - result = self._get_single_result(i, task_id, token_id, task) - - self.tokens_counter[task_id] += 1 - if token_id not in task["eos_token_ids"]: - self.all_tokens[i].append(token_id) - - self.number_of_output_tokens += 1 - if token_id in task["eos_token_ids"]: - self._recycle_resources(task_id, i, task) - model_server_logger.info("req_id: {0} finished".format(task_id)) - model_server_logger.info(f"{self.resource_manager.info()}") - exist_finished_task = True + result = self._get_single_result(i, task_id, token_ids, task) + + for token_id in token_ids: + self.tokens_counter[task_id] += 1 + if token_id not in task["eos_token_ids"]: + self.all_tokens[i].append(token_id) + + self.number_of_output_tokens += 1 + if token_id in task["eos_token_ids"]: + self._recycle_resources(task_id, i, task) + model_server_logger.info("req_id: {0} finished".format(task_id)) + model_server_logger.info(f"{self.resource_manager.info()}") + exist_finished_task = True + break batch_result.append(result) self.postprocess(batch_result, exist_finished_task) @@ -228,7 +251,10 @@ def process_sampling_results(self): while self._is_running: try: rank_id = 0 - get_output(self.output_tokens, rank_id, self._is_blocking) + if self.is_speculate_decoding: + speculate_get_output(self.output_tokens, rank_id, self._is_blocking) + else: + get_output(self.output_tokens, rank_id, self._is_blocking) if self.output_tokens[0, 0] == -2: continue diff --git a/llm/server/server/http_server/__init__.py b/services/llm/server/server/http_server/__init__.py similarity index 100% rename from llm/server/server/http_server/__init__.py rename to services/llm/server/server/http_server/__init__.py diff --git a/llm/server/server/http_server/adapter_openai.py b/services/llm/server/server/http_server/adapter_openai.py similarity index 100% rename from llm/server/server/http_server/adapter_openai.py rename to services/llm/server/server/http_server/adapter_openai.py diff --git a/llm/server/server/http_server/api.py b/services/llm/server/server/http_server/api.py similarity index 99% rename from llm/server/server/http_server/api.py rename to services/llm/server/server/http_server/api.py index df9c066284f..2e01ae039db 100644 --- a/llm/server/server/http_server/api.py +++ b/services/llm/server/server/http_server/api.py @@ -31,6 +31,7 @@ class Req(BaseModel): req_id: str = Field(default_factory=lambda: str(uuid.uuid4())) input_ids: Optional[List[int]] = None text: Optional[str] = None + stop_sequences: Optional[List] = None messages: Optional[List] = None max_dec_len: Optional[int] = None seq_len: Optional[int] = None diff --git a/llm/server/server/http_server/app.py b/services/llm/server/server/http_server/app.py similarity index 100% rename from llm/server/server/http_server/app.py rename to services/llm/server/server/http_server/app.py diff --git a/llm/server/server/triton_server.py b/services/llm/server/server/triton_server.py similarity index 93% rename from llm/server/server/triton_server.py rename to services/llm/server/server/triton_server.py index 12024c251af..02be0b4e8aa 100644 --- a/llm/server/server/triton_server.py +++ b/services/llm/server/server/triton_server.py @@ -98,11 +98,35 @@ def _push_mode_sender_thread(self): except Exception as e: model_server_logger.error("Unexcepted error happend: {}, {}".format(e, str(traceback.format_exc()))) + def _cache_special_tokens(self, batch_result): + for i in range(len(batch_result)): + is_end = batch_result[i].get("is_end", 0) + token_ids = batch_result[i]["token_ids"] + if is_end != 1: + if batch_result[i]["req_id"] not in self.token_buffer: + self.token_buffer[batch_result[i]["req_id"]] = list() + self.score_buffer[batch_result[i]["req_id"]] = list() + self.token_buffer[batch_result[i]["req_id"]].extend(token_ids) + self.score_buffer[batch_result[i]["req_id"]].extend(batch_result[i].get("token_scores", [])) + batch_result[i]["token_ids"] = [] + if "token_scores" in batch_result[i]: + batch_result[i]["token_scores"] = [] + else: + if batch_result[i]["req_id"] in self.token_buffer: + batch_result[i]["token_ids"] = self.token_buffer[batch_result[i] + ["req_id"]] + batch_result[i]["token_ids"] + del self.token_buffer[batch_result[i]["req_id"]] + if "token_scores" in batch_result[i]: + batch_result[i]["token_scores"] = self.score_buffer[batch_result[i] + ["req_id"]] + batch_result[i]["token_scores"] + del self.score_buffer[batch_result[i]["req_id"]] + def postprocess(self, batch_result, exist_finished_task=False): """ single postprocess for triton """ try: + self._cache_special_tokens(batch_result) self.cached_generated_tokens.put(batch_result) except Exception as e: model_server_logger.info( diff --git a/llm/server/server/triton_server_helper.py b/services/llm/server/server/triton_server_helper.py similarity index 100% rename from llm/server/server/triton_server_helper.py rename to services/llm/server/server/triton_server_helper.py diff --git a/llm/server/server/utils.py b/services/llm/server/server/utils.py similarity index 100% rename from llm/server/server/utils.py rename to services/llm/server/server/utils.py diff --git a/serving/CMakeLists.txt b/services/serving/CMakeLists.txt similarity index 100% rename from serving/CMakeLists.txt rename to services/serving/CMakeLists.txt diff --git a/serving/Dockerfile b/services/serving/Dockerfile similarity index 100% rename from serving/Dockerfile rename to services/serving/Dockerfile diff --git a/serving/Dockerfile_CUDA_11_2 b/services/serving/Dockerfile_CUDA_11_2 similarity index 100% rename from serving/Dockerfile_CUDA_11_2 rename to services/serving/Dockerfile_CUDA_11_2 diff --git a/serving/Dockerfile_CUDA_11_2_TRT_8_5_PADDLE_2_4_2 b/services/serving/Dockerfile_CUDA_11_2_TRT_8_5_PADDLE_2_4_2 similarity index 100% rename from serving/Dockerfile_CUDA_11_2_TRT_8_5_PADDLE_2_4_2 rename to services/serving/Dockerfile_CUDA_11_2_TRT_8_5_PADDLE_2_4_2 diff --git a/serving/Dockerfile_CUDA_11_4_TRT_8_4 b/services/serving/Dockerfile_CUDA_11_4_TRT_8_4 similarity index 100% rename from serving/Dockerfile_CUDA_11_4_TRT_8_4 rename to services/serving/Dockerfile_CUDA_11_4_TRT_8_4 diff --git a/serving/Dockerfile_cpu b/services/serving/Dockerfile_cpu similarity index 100% rename from serving/Dockerfile_cpu rename to services/serving/Dockerfile_cpu diff --git a/serving/Dockerfile_ipu b/services/serving/Dockerfile_ipu similarity index 100% rename from serving/Dockerfile_ipu rename to services/serving/Dockerfile_ipu diff --git a/serving/Dockerfile_xpu b/services/serving/Dockerfile_xpu similarity index 100% rename from serving/Dockerfile_xpu rename to services/serving/Dockerfile_xpu diff --git a/serving/Dockerfile_xpu_encrypt_auth b/services/serving/Dockerfile_xpu_encrypt_auth similarity index 100% rename from serving/Dockerfile_xpu_encrypt_auth rename to services/serving/Dockerfile_xpu_encrypt_auth diff --git a/serving/README.md b/services/serving/README.md similarity index 92% rename from serving/README.md rename to services/serving/README.md index 0e7d2e13fd9..ea5ce1223d5 100644 --- a/serving/README.md +++ b/services/serving/README.md @@ -37,6 +37,8 @@ Users can also compile the image by themselves according to their own needs, ref - [FastDeploy Serving Deployment Image Compilation](docs/EN/compile-en.md) +> **Note:** The proxy settings have been pre-configured for this image. If these settings are not required for your environment, you can disable the proxy by executing the commands `unset https_proxy` and `unset http_proxy`. + ## Other Tutorials - [How to Prepare Serving Model Repository](docs/EN/model_repository-en.md) diff --git a/serving/README_CN.md b/services/serving/README_CN.md similarity index 94% rename from serving/README_CN.md rename to services/serving/README_CN.md index ab0245b1db2..b51cba0333b 100644 --- a/serving/README_CN.md +++ b/services/serving/README_CN.md @@ -30,6 +30,8 @@ docker pull registry.baidubce.com/paddlepaddle/fastdeploy:1.0.7-gpu-cuda11.4-trt 用户也可根据自身需求,参考如下文档自行编译镜像 - [FastDeploy服务化部署镜像编译说明](docs/zh_CN/compile.md) +> **注意:** 该镜像已预配置代理。如果您的环境不需要代理,可以通过执行命令`unset https_proxy`和`unset http_proxy`来禁用代理。 + ## 其它文档 - [模型仓库目录说明](docs/zh_CN/model_repository.md) (说明如何准备模型仓库目录) - [模型配置说明](docs/zh_CN/model_configuration.md) (说明runtime的配置选项) diff --git a/serving/docs/EN/client-en.md b/services/serving/docs/EN/client-en.md similarity index 100% rename from serving/docs/EN/client-en.md rename to services/serving/docs/EN/client-en.md diff --git a/serving/docs/EN/compile-en.md b/services/serving/docs/EN/compile-en.md similarity index 100% rename from serving/docs/EN/compile-en.md rename to services/serving/docs/EN/compile-en.md diff --git a/serving/docs/EN/compile_without_docker_centos-en.md b/services/serving/docs/EN/compile_without_docker_centos-en.md similarity index 100% rename from serving/docs/EN/compile_without_docker_centos-en.md rename to services/serving/docs/EN/compile_without_docker_centos-en.md diff --git a/serving/docs/EN/demo-en.md b/services/serving/docs/EN/demo-en.md similarity index 100% rename from serving/docs/EN/demo-en.md rename to services/serving/docs/EN/demo-en.md diff --git a/serving/docs/EN/model_configuration-en.md b/services/serving/docs/EN/model_configuration-en.md similarity index 100% rename from serving/docs/EN/model_configuration-en.md rename to services/serving/docs/EN/model_configuration-en.md diff --git a/serving/docs/EN/model_repository-en.md b/services/serving/docs/EN/model_repository-en.md similarity index 100% rename from serving/docs/EN/model_repository-en.md rename to services/serving/docs/EN/model_repository-en.md diff --git a/serving/docs/EN/vdl_management-en.md b/services/serving/docs/EN/vdl_management-en.md similarity index 100% rename from serving/docs/EN/vdl_management-en.md rename to services/serving/docs/EN/vdl_management-en.md diff --git a/serving/docs/zh_CN/client.md b/services/serving/docs/zh_CN/client.md similarity index 100% rename from serving/docs/zh_CN/client.md rename to services/serving/docs/zh_CN/client.md diff --git a/serving/docs/zh_CN/compile.md b/services/serving/docs/zh_CN/compile.md similarity index 100% rename from serving/docs/zh_CN/compile.md rename to services/serving/docs/zh_CN/compile.md diff --git a/serving/docs/zh_CN/compile_without_docker_centos.md b/services/serving/docs/zh_CN/compile_without_docker_centos.md similarity index 100% rename from serving/docs/zh_CN/compile_without_docker_centos.md rename to services/serving/docs/zh_CN/compile_without_docker_centos.md diff --git a/serving/docs/zh_CN/demo.md b/services/serving/docs/zh_CN/demo.md similarity index 100% rename from serving/docs/zh_CN/demo.md rename to services/serving/docs/zh_CN/demo.md diff --git a/serving/docs/zh_CN/model_configuration.md b/services/serving/docs/zh_CN/model_configuration.md similarity index 100% rename from serving/docs/zh_CN/model_configuration.md rename to services/serving/docs/zh_CN/model_configuration.md diff --git a/serving/docs/zh_CN/model_repository.md b/services/serving/docs/zh_CN/model_repository.md similarity index 100% rename from serving/docs/zh_CN/model_repository.md rename to services/serving/docs/zh_CN/model_repository.md diff --git a/serving/docs/zh_CN/vdl_management.md b/services/serving/docs/zh_CN/vdl_management.md similarity index 100% rename from serving/docs/zh_CN/vdl_management.md rename to services/serving/docs/zh_CN/vdl_management.md diff --git a/serving/docs/zh_CN/xpu.md b/services/serving/docs/zh_CN/xpu.md similarity index 100% rename from serving/docs/zh_CN/xpu.md rename to services/serving/docs/zh_CN/xpu.md diff --git a/serving/scripts/build.sh b/services/serving/scripts/build.sh similarity index 100% rename from serving/scripts/build.sh rename to services/serving/scripts/build.sh diff --git a/serving/scripts/build_fd_cuda_11_2.sh b/services/serving/scripts/build_fd_cuda_11_2.sh similarity index 100% rename from serving/scripts/build_fd_cuda_11_2.sh rename to services/serving/scripts/build_fd_cuda_11_2.sh diff --git a/serving/scripts/build_fd_cuda_11_2_trt_8_5_paddle_2_4_2.sh b/services/serving/scripts/build_fd_cuda_11_2_trt_8_5_paddle_2_4_2.sh similarity index 100% rename from serving/scripts/build_fd_cuda_11_2_trt_8_5_paddle_2_4_2.sh rename to services/serving/scripts/build_fd_cuda_11_2_trt_8_5_paddle_2_4_2.sh diff --git a/serving/scripts/build_fd_ipu.sh b/services/serving/scripts/build_fd_ipu.sh similarity index 100% rename from serving/scripts/build_fd_ipu.sh rename to services/serving/scripts/build_fd_ipu.sh diff --git a/serving/scripts/build_fd_xpu.sh b/services/serving/scripts/build_fd_xpu.sh similarity index 100% rename from serving/scripts/build_fd_xpu.sh rename to services/serving/scripts/build_fd_xpu.sh diff --git a/serving/scripts/build_fd_xpu_encrypt_auth.sh b/services/serving/scripts/build_fd_xpu_encrypt_auth.sh similarity index 100% rename from serving/scripts/build_fd_xpu_encrypt_auth.sh rename to services/serving/scripts/build_fd_xpu_encrypt_auth.sh diff --git a/serving/scripts/build_triton_fd_backend.sh b/services/serving/scripts/build_triton_fd_backend.sh similarity index 100% rename from serving/scripts/build_triton_fd_backend.sh rename to services/serving/scripts/build_triton_fd_backend.sh diff --git a/serving/src/fastdeploy_backend_utils.cc b/services/serving/src/fastdeploy_backend_utils.cc similarity index 100% rename from serving/src/fastdeploy_backend_utils.cc rename to services/serving/src/fastdeploy_backend_utils.cc diff --git a/serving/src/fastdeploy_backend_utils.h b/services/serving/src/fastdeploy_backend_utils.h similarity index 100% rename from serving/src/fastdeploy_backend_utils.h rename to services/serving/src/fastdeploy_backend_utils.h diff --git a/serving/src/fastdeploy_runtime.cc b/services/serving/src/fastdeploy_runtime.cc similarity index 100% rename from serving/src/fastdeploy_runtime.cc rename to services/serving/src/fastdeploy_runtime.cc diff --git a/serving/src/libtriton_fastdeploy.ldscript b/services/serving/src/libtriton_fastdeploy.ldscript similarity index 100% rename from serving/src/libtriton_fastdeploy.ldscript rename to services/serving/src/libtriton_fastdeploy.ldscript diff --git a/streamer/CMakeLists.txt b/services/streamer/CMakeLists.txt similarity index 100% rename from streamer/CMakeLists.txt rename to services/streamer/CMakeLists.txt diff --git a/streamer/README.md b/services/streamer/README.md similarity index 100% rename from streamer/README.md rename to services/streamer/README.md diff --git a/streamer/README_CN.md b/services/streamer/README_CN.md similarity index 100% rename from streamer/README_CN.md rename to services/streamer/README_CN.md diff --git a/streamer/README_EN.md b/services/streamer/README_EN.md similarity index 100% rename from streamer/README_EN.md rename to services/streamer/README_EN.md diff --git a/streamer/docs/cn/yaml_config.md b/services/streamer/docs/cn/yaml_config.md similarity index 100% rename from streamer/docs/cn/yaml_config.md rename to services/streamer/docs/cn/yaml_config.md diff --git a/streamer/docs/en/yaml_config.md b/services/streamer/docs/en/yaml_config.md similarity index 100% rename from streamer/docs/en/yaml_config.md rename to services/streamer/docs/en/yaml_config.md diff --git a/streamer/examples/ppyoloe/cpp/CMakeLists.txt b/services/streamer/examples/ppyoloe/cpp/CMakeLists.txt similarity index 100% rename from streamer/examples/ppyoloe/cpp/CMakeLists.txt rename to services/streamer/examples/ppyoloe/cpp/CMakeLists.txt diff --git a/streamer/examples/ppyoloe/cpp/README.md b/services/streamer/examples/ppyoloe/cpp/README.md similarity index 100% rename from streamer/examples/ppyoloe/cpp/README.md rename to services/streamer/examples/ppyoloe/cpp/README.md diff --git a/streamer/examples/ppyoloe/cpp/README_CN.md b/services/streamer/examples/ppyoloe/cpp/README_CN.md similarity index 100% rename from streamer/examples/ppyoloe/cpp/README_CN.md rename to services/streamer/examples/ppyoloe/cpp/README_CN.md diff --git a/streamer/examples/ppyoloe/cpp/README_EN.md b/services/streamer/examples/ppyoloe/cpp/README_EN.md similarity index 100% rename from streamer/examples/ppyoloe/cpp/README_EN.md rename to services/streamer/examples/ppyoloe/cpp/README_EN.md diff --git a/streamer/examples/ppyoloe/cpp/main.cc b/services/streamer/examples/ppyoloe/cpp/main.cc similarity index 100% rename from streamer/examples/ppyoloe/cpp/main.cc rename to services/streamer/examples/ppyoloe/cpp/main.cc diff --git a/streamer/examples/ppyoloe/cpp/nvinfer_config.txt b/services/streamer/examples/ppyoloe/cpp/nvinfer_config.txt similarity index 100% rename from streamer/examples/ppyoloe/cpp/nvinfer_config.txt rename to services/streamer/examples/ppyoloe/cpp/nvinfer_config.txt diff --git a/streamer/examples/ppyoloe/cpp/streamer_cfg.yml b/services/streamer/examples/ppyoloe/cpp/streamer_cfg.yml similarity index 100% rename from streamer/examples/ppyoloe/cpp/streamer_cfg.yml rename to services/streamer/examples/ppyoloe/cpp/streamer_cfg.yml diff --git a/streamer/examples/ppyoloe/python/main.py b/services/streamer/examples/ppyoloe/python/main.py similarity index 100% rename from streamer/examples/ppyoloe/python/main.py rename to services/streamer/examples/ppyoloe/python/main.py diff --git a/streamer/examples/video_decoder/cpp/CMakeLists.txt b/services/streamer/examples/video_decoder/cpp/CMakeLists.txt similarity index 100% rename from streamer/examples/video_decoder/cpp/CMakeLists.txt rename to services/streamer/examples/video_decoder/cpp/CMakeLists.txt diff --git a/streamer/examples/video_decoder/cpp/README.md b/services/streamer/examples/video_decoder/cpp/README.md similarity index 100% rename from streamer/examples/video_decoder/cpp/README.md rename to services/streamer/examples/video_decoder/cpp/README.md diff --git a/streamer/examples/video_decoder/cpp/README_CN.md b/services/streamer/examples/video_decoder/cpp/README_CN.md similarity index 100% rename from streamer/examples/video_decoder/cpp/README_CN.md rename to services/streamer/examples/video_decoder/cpp/README_CN.md diff --git a/streamer/examples/video_decoder/cpp/README_EN.md b/services/streamer/examples/video_decoder/cpp/README_EN.md similarity index 100% rename from streamer/examples/video_decoder/cpp/README_EN.md rename to services/streamer/examples/video_decoder/cpp/README_EN.md diff --git a/streamer/examples/video_decoder/cpp/cpu.yml b/services/streamer/examples/video_decoder/cpp/cpu.yml similarity index 100% rename from streamer/examples/video_decoder/cpp/cpu.yml rename to services/streamer/examples/video_decoder/cpp/cpu.yml diff --git a/streamer/examples/video_decoder/cpp/gpu.yml b/services/streamer/examples/video_decoder/cpp/gpu.yml similarity index 100% rename from streamer/examples/video_decoder/cpp/gpu.yml rename to services/streamer/examples/video_decoder/cpp/gpu.yml diff --git a/streamer/examples/video_decoder/cpp/main.cc b/services/streamer/examples/video_decoder/cpp/main.cc similarity index 100% rename from streamer/examples/video_decoder/cpp/main.cc rename to services/streamer/examples/video_decoder/cpp/main.cc diff --git a/streamer/python/scripts/__init__.py b/services/streamer/python/scripts/__init__.py similarity index 100% rename from streamer/python/scripts/__init__.py rename to services/streamer/python/scripts/__init__.py diff --git a/streamer/python/scripts/process_libraries.py b/services/streamer/python/scripts/process_libraries.py similarity index 100% rename from streamer/python/scripts/process_libraries.py rename to services/streamer/python/scripts/process_libraries.py diff --git a/streamer/python/setup.py b/services/streamer/python/setup.py similarity index 97% rename from streamer/python/setup.py rename to services/streamer/python/setup.py index 5e4129f0f64..bcc0604a6c4 100644 --- a/streamer/python/setup.py +++ b/services/streamer/python/setup.py @@ -70,9 +70,13 @@ except (OSError, subprocess.CalledProcessError): git_version = None -with open(os.path.join(TOP_DIR, '..', 'VERSION_NUMBER')) as version_file: - VersionInfo = namedtuple('VersionInfo', ['version', 'git_version'])( - version=version_file.read().strip(), git_version=git_version) +version_str = os.getenv("FASTDEPLOY_VERSION", "0.0.0") +version_file_path = os.path.join(TOP_DIR, '..', '..', 'VERSION_NUMBER') +if os.path.exists(version_file_path): + with open(version_file_path) as vf: + version_str = vf.read().strip() +VersionInfo = namedtuple('VersionInfo', ['version', 'git_version'])( + version=version_str, git_version=git_version) ################################################################################ # Pre Check diff --git a/streamer/python/streamer/__init__.py b/services/streamer/python/streamer/__init__.py similarity index 100% rename from streamer/python/streamer/__init__.py rename to services/streamer/python/streamer/__init__.py diff --git a/streamer/python/streamer/c_lib_wrap.py b/services/streamer/python/streamer/c_lib_wrap.py similarity index 100% rename from streamer/python/streamer/c_lib_wrap.py rename to services/streamer/python/streamer/c_lib_wrap.py diff --git a/streamer/python/streamer/fd_streamer.py b/services/streamer/python/streamer/fd_streamer.py similarity index 100% rename from streamer/python/streamer/fd_streamer.py rename to services/streamer/python/streamer/fd_streamer.py diff --git a/streamer/src/app/base_app.cc b/services/streamer/src/app/base_app.cc similarity index 100% rename from streamer/src/app/base_app.cc rename to services/streamer/src/app/base_app.cc diff --git a/streamer/src/app/base_app.h b/services/streamer/src/app/base_app.h similarity index 100% rename from streamer/src/app/base_app.h rename to services/streamer/src/app/base_app.h diff --git a/streamer/src/app/video_analytics.cc b/services/streamer/src/app/video_analytics.cc similarity index 100% rename from streamer/src/app/video_analytics.cc rename to services/streamer/src/app/video_analytics.cc diff --git a/streamer/src/app/video_analytics.h b/services/streamer/src/app/video_analytics.h similarity index 100% rename from streamer/src/app/video_analytics.h rename to services/streamer/src/app/video_analytics.h diff --git a/streamer/src/app/video_decoder.cc b/services/streamer/src/app/video_decoder.cc similarity index 100% rename from streamer/src/app/video_decoder.cc rename to services/streamer/src/app/video_decoder.cc diff --git a/streamer/src/app/video_decoder.h b/services/streamer/src/app/video_decoder.h similarity index 100% rename from streamer/src/app/video_decoder.h rename to services/streamer/src/app/video_decoder.h diff --git a/streamer/src/app/yaml_parser.cc b/services/streamer/src/app/yaml_parser.cc similarity index 100% rename from streamer/src/app/yaml_parser.cc rename to services/streamer/src/app/yaml_parser.cc diff --git a/streamer/src/app/yaml_parser.h b/services/streamer/src/app/yaml_parser.h similarity index 100% rename from streamer/src/app/yaml_parser.h rename to services/streamer/src/app/yaml_parser.h diff --git a/streamer/src/deepstream/bbox_parser.cc b/services/streamer/src/deepstream/bbox_parser.cc similarity index 100% rename from streamer/src/deepstream/bbox_parser.cc rename to services/streamer/src/deepstream/bbox_parser.cc diff --git a/streamer/src/fd_streamer.cc b/services/streamer/src/fd_streamer.cc similarity index 100% rename from streamer/src/fd_streamer.cc rename to services/streamer/src/fd_streamer.cc diff --git a/streamer/src/fd_streamer.h b/services/streamer/src/fd_streamer.h similarity index 100% rename from streamer/src/fd_streamer.h rename to services/streamer/src/fd_streamer.h diff --git a/streamer/src/fd_streamer_pybind.cc b/services/streamer/src/fd_streamer_pybind.cc similarity index 100% rename from streamer/src/fd_streamer_pybind.cc rename to services/streamer/src/fd_streamer_pybind.cc diff --git a/streamer/src/gstreamer/meta/CMakeLists.txt b/services/streamer/src/gstreamer/meta/CMakeLists.txt similarity index 100% rename from streamer/src/gstreamer/meta/CMakeLists.txt rename to services/streamer/src/gstreamer/meta/CMakeLists.txt diff --git a/streamer/src/gstreamer/meta/meta.cc b/services/streamer/src/gstreamer/meta/meta.cc similarity index 100% rename from streamer/src/gstreamer/meta/meta.cc rename to services/streamer/src/gstreamer/meta/meta.cc diff --git a/streamer/src/gstreamer/meta/meta.h b/services/streamer/src/gstreamer/meta/meta.h similarity index 100% rename from streamer/src/gstreamer/meta/meta.h rename to services/streamer/src/gstreamer/meta/meta.h diff --git a/streamer/src/gstreamer/perf.cc b/services/streamer/src/gstreamer/perf.cc similarity index 100% rename from streamer/src/gstreamer/perf.cc rename to services/streamer/src/gstreamer/perf.cc diff --git a/streamer/src/gstreamer/perf.h b/services/streamer/src/gstreamer/perf.h similarity index 100% rename from streamer/src/gstreamer/perf.h rename to services/streamer/src/gstreamer/perf.h diff --git a/streamer/src/gstreamer/plugin/fdinfer/CMakeLists.txt b/services/streamer/src/gstreamer/plugin/fdinfer/CMakeLists.txt similarity index 100% rename from streamer/src/gstreamer/plugin/fdinfer/CMakeLists.txt rename to services/streamer/src/gstreamer/plugin/fdinfer/CMakeLists.txt diff --git a/streamer/src/gstreamer/plugin/fdinfer/fdinfer.cc b/services/streamer/src/gstreamer/plugin/fdinfer/fdinfer.cc similarity index 100% rename from streamer/src/gstreamer/plugin/fdinfer/fdinfer.cc rename to services/streamer/src/gstreamer/plugin/fdinfer/fdinfer.cc diff --git a/streamer/src/gstreamer/plugin/fdinfer/fdinfer.h b/services/streamer/src/gstreamer/plugin/fdinfer/fdinfer.h similarity index 100% rename from streamer/src/gstreamer/plugin/fdinfer/fdinfer.h rename to services/streamer/src/gstreamer/plugin/fdinfer/fdinfer.h diff --git a/streamer/src/gstreamer/plugin/fdinfer/fdmodel.cc b/services/streamer/src/gstreamer/plugin/fdinfer/fdmodel.cc similarity index 100% rename from streamer/src/gstreamer/plugin/fdinfer/fdmodel.cc rename to services/streamer/src/gstreamer/plugin/fdinfer/fdmodel.cc diff --git a/streamer/src/gstreamer/plugin/fdinfer/fdmodel.h b/services/streamer/src/gstreamer/plugin/fdinfer/fdmodel.h similarity index 100% rename from streamer/src/gstreamer/plugin/fdinfer/fdmodel.h rename to services/streamer/src/gstreamer/plugin/fdinfer/fdmodel.h diff --git a/streamer/src/gstreamer/plugin/fdtracker/CMakeLists.txt b/services/streamer/src/gstreamer/plugin/fdtracker/CMakeLists.txt similarity index 100% rename from streamer/src/gstreamer/plugin/fdtracker/CMakeLists.txt rename to services/streamer/src/gstreamer/plugin/fdtracker/CMakeLists.txt diff --git a/streamer/src/gstreamer/plugin/fdtracker/gstfdtracker.cc b/services/streamer/src/gstreamer/plugin/fdtracker/gstfdtracker.cc similarity index 100% rename from streamer/src/gstreamer/plugin/fdtracker/gstfdtracker.cc rename to services/streamer/src/gstreamer/plugin/fdtracker/gstfdtracker.cc diff --git a/streamer/src/gstreamer/plugin/fdtracker/gstfdtracker.h b/services/streamer/src/gstreamer/plugin/fdtracker/gstfdtracker.h similarity index 100% rename from streamer/src/gstreamer/plugin/fdtracker/gstfdtracker.h rename to services/streamer/src/gstreamer/plugin/fdtracker/gstfdtracker.h diff --git a/streamer/src/gstreamer/plugin/fdtracker/include/kalmantracker.h b/services/streamer/src/gstreamer/plugin/fdtracker/include/kalmantracker.h similarity index 100% rename from streamer/src/gstreamer/plugin/fdtracker/include/kalmantracker.h rename to services/streamer/src/gstreamer/plugin/fdtracker/include/kalmantracker.h diff --git a/streamer/src/gstreamer/plugin/fdtracker/include/lapjv.h b/services/streamer/src/gstreamer/plugin/fdtracker/include/lapjv.h similarity index 100% rename from streamer/src/gstreamer/plugin/fdtracker/include/lapjv.h rename to services/streamer/src/gstreamer/plugin/fdtracker/include/lapjv.h diff --git a/streamer/src/gstreamer/plugin/fdtracker/include/ocsort.h b/services/streamer/src/gstreamer/plugin/fdtracker/include/ocsort.h similarity index 100% rename from streamer/src/gstreamer/plugin/fdtracker/include/ocsort.h rename to services/streamer/src/gstreamer/plugin/fdtracker/include/ocsort.h diff --git a/streamer/src/gstreamer/plugin/fdtracker/include/trajectory.h b/services/streamer/src/gstreamer/plugin/fdtracker/include/trajectory.h similarity index 100% rename from streamer/src/gstreamer/plugin/fdtracker/include/trajectory.h rename to services/streamer/src/gstreamer/plugin/fdtracker/include/trajectory.h diff --git a/streamer/src/gstreamer/plugin/fdtracker/src/kalmantracker.cpp b/services/streamer/src/gstreamer/plugin/fdtracker/src/kalmantracker.cpp similarity index 100% rename from streamer/src/gstreamer/plugin/fdtracker/src/kalmantracker.cpp rename to services/streamer/src/gstreamer/plugin/fdtracker/src/kalmantracker.cpp diff --git a/streamer/src/gstreamer/plugin/fdtracker/src/lapjv.cpp b/services/streamer/src/gstreamer/plugin/fdtracker/src/lapjv.cpp similarity index 100% rename from streamer/src/gstreamer/plugin/fdtracker/src/lapjv.cpp rename to services/streamer/src/gstreamer/plugin/fdtracker/src/lapjv.cpp diff --git a/streamer/src/gstreamer/plugin/fdtracker/src/ocsort.cpp b/services/streamer/src/gstreamer/plugin/fdtracker/src/ocsort.cpp similarity index 100% rename from streamer/src/gstreamer/plugin/fdtracker/src/ocsort.cpp rename to services/streamer/src/gstreamer/plugin/fdtracker/src/ocsort.cpp diff --git a/streamer/src/gstreamer/plugin/fdtracker/src/trajectory.cpp b/services/streamer/src/gstreamer/plugin/fdtracker/src/trajectory.cpp similarity index 100% rename from streamer/src/gstreamer/plugin/fdtracker/src/trajectory.cpp rename to services/streamer/src/gstreamer/plugin/fdtracker/src/trajectory.cpp diff --git a/streamer/src/gstreamer/types.h b/services/streamer/src/gstreamer/types.h similarity index 100% rename from streamer/src/gstreamer/types.h rename to services/streamer/src/gstreamer/types.h diff --git a/streamer/src/gstreamer/utils.cc b/services/streamer/src/gstreamer/utils.cc similarity index 100% rename from streamer/src/gstreamer/utils.cc rename to services/streamer/src/gstreamer/utils.cc diff --git a/streamer/src/gstreamer/utils.h b/services/streamer/src/gstreamer/utils.h similarity index 100% rename from streamer/src/gstreamer/utils.h rename to services/streamer/src/gstreamer/utils.h diff --git a/streamer/src/pybind/main.cc b/services/streamer/src/pybind/main.cc similarity index 100% rename from streamer/src/pybind/main.cc rename to services/streamer/src/pybind/main.cc diff --git a/streamer/src/pybind/main.h b/services/streamer/src/pybind/main.h similarity index 100% rename from streamer/src/pybind/main.h rename to services/streamer/src/pybind/main.h diff --git a/tools/extract_ppocr_dict.py b/tools/extract_ppocr_dict.py new file mode 100644 index 00000000000..ca4cf25fcdd --- /dev/null +++ b/tools/extract_ppocr_dict.py @@ -0,0 +1,131 @@ +#!/usr/bin/env python3 +# -*- coding: utf-8 -*- + +# Copyright (c) 2024 PaddlePaddle Authors. All Rights Reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import argparse +import json +import os +import sys + + +def unquote_yaml_scalar(val): + if len(val) >= 2 and val.startswith("'") and val.endswith("'"): + # YAML single quotes: escaped quote is '' + return val[1:-1].replace("''", "'") + if len(val) >= 2 and val.startswith('"') and val.endswith('"'): + # YAML double quotes: standard JSON/C-style escape + try: + return json.loads(val) + except Exception: + return val[1:-1] + return val + + +def extract_dict(yaml_file, output_file): + if not os.path.exists(yaml_file): + raise FileNotFoundError(f"YAML file not found: {yaml_file}") + + characters = [] + + # Try PyYAML first if installed + try: + import yaml + with open(yaml_file, "r", encoding="utf-8") as f: + data = yaml.safe_load(f) + if isinstance(data, dict): + # Check PostProcess -> character_dict + post_process = data.get("PostProcess", {}) + if isinstance(post_process, dict) and "character_dict" in post_process: + characters = post_process["character_dict"] + elif "character_dict" in data: + characters = data["character_dict"] + except ImportError: + pass + + # Fallback to pure-Python YAML list parser if PyYAML unavailable or not found + if not characters: + with open(yaml_file, "r", encoding="utf-8") as f: + lines = f.readlines() + + in_character_dict = False + dict_indent = None + + for line in lines: + line_str = line.rstrip("\r\n") + # Only strip ASCII space/tab indentation from left to preserve Unicode characters + lstripped = line_str.lstrip(" \t") + + if lstripped.startswith("character_dict:"): + in_character_dict = True + dict_indent = len(line_str) - len(lstripped) + continue + + if in_character_dict: + if not lstripped: + continue + current_indent = len(line_str) - len(lstripped) + if current_indent <= dict_indent and not lstripped.startswith("-"): + # Exited character_dict section + break + + # Check if list item under character_dict + if lstripped.startswith("- "): + raw_val = lstripped[2:] + char_val = unquote_yaml_scalar(raw_val) + characters.append(char_val) + elif lstripped == "-": + characters.append("") + + if not characters: + raise ValueError(f"No character_dict found in {yaml_file}") + + output_dir = os.path.dirname(output_file) + if output_dir and not os.path.exists(output_dir): + os.makedirs(output_dir, exist_ok=True) + + with open(output_file, "w", encoding="utf-8") as f: + for ch in characters: + f.write(f"{ch}\n") + + print(f"Successfully extracted {len(characters)} characters to: {output_file}") + if characters: + print(f" First 5 items: {characters[:5]}") + print(f" Last 5 items: {characters[-5:]}") + + +def main(): + parser = argparse.ArgumentParser( + description="Extract character dictionary from PP-OCRv5/v6 inference.yml" + ) + parser.add_argument( + "--yaml_file", + "-y", + required=True, + help="Path to inference.yml containing character_dict", + ) + parser.add_argument( + "--output_file", + "-o", + default="ppocr_keys.txt", + help="Path to output dictionary text file (default: ppocr_keys.txt)", + ) + args = parser.parse_args() + + extract_dict(args.yaml_file, args.output_file) + + +if __name__ == "__main__": + main()