diff --git a/CMakeLists.txt b/CMakeLists.txt index f662b8c22e5..17bf24485d6 100755 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -8,6 +8,20 @@ find_package(onnxruntime REQUIRED) find_package(Eigen3 REQUIRED) message("OpenCV_LIBS: ${OpenCV_LIBS} ${OPENCV_core_FOUND} ${OPENCV_WORLD_FOUND}") +# ONNX Runtime flattens the headers of every enabled execution provider into +# , so exists exactly when +# that ONNX Runtime was built with the WebGPU EP (the same signal the Maa +# side uses for DirectML/CoreML). FastDeploy always compiles the WebGPU device +# path and falls back to Device::CPU at runtime when the EP is unavailable, so +# this reports what the linked ONNX Runtime can actually do -- pick the ONNX +# Runtime accordingly if you need the device. +if(EXISTS "${onnxruntime_INCLUDE_DIR}/webgpu_provider_factory.h") + set(WITH_WEBGPU ON) +else() + set(WITH_WEBGPU OFF) +endif() +message(STATUS "WITH_WEBGPU: ${WITH_WEBGPU} (onnxruntime_INCLUDE_DIR=${onnxruntime_INCLUDE_DIR})") + option(WITH_CUDA "Whether WITH_CUDA=ON, will enable onnxruntime-gpu/paddle-inference-gpu" OFF) option(PRINT_INFO "Print more debug info while running" OFF) @@ -117,7 +131,11 @@ set_target_properties( target_include_directories(fastdeploy_ppocr INTERFACE $ # for build - $ # for install + # Headers are installed to /include (see the install() below). + # CMAKE_INSTALL_INCLUDE is not defined in this project, so spell the + # relative path out: an empty INSTALL_INTERFACE leaves the imported target + # without any include directory for consumers. + $ # for install ) target_link_libraries(fastdeploy_ppocr PUBLIC ${OpenCV_LIBS} PRIVATE onnxruntime::onnxruntime) @@ -129,7 +147,27 @@ if(ANDROID) endif() install(TARGETS fastdeploy_ppocr EXPORT fastdeploy_ppocrConfig) -install(EXPORT fastdeploy_ppocrConfig DESTINATION share/fastdeploy_ppocr) +include(CMakePackageConfigHelpers) +install(EXPORT fastdeploy_ppocrConfig + DESTINATION share/fastdeploy_ppocr + FILE fastdeploy_ppocrTargets.cmake) +# A static build leaks $ into the exported +# interface, so the package config has to help consumers materialize that +# target; ship the Find module it needs. +get_target_property(FASTDEPLOY_PPOCR_LIBRARY_TYPE fastdeploy_ppocr TYPE) +if(FASTDEPLOY_PPOCR_LIBRARY_TYPE STREQUAL "STATIC_LIBRARY") + set(FASTDEPLOY_PPOCR_STATIC ON) +else() + set(FASTDEPLOY_PPOCR_STATIC OFF) +endif() +install(FILES cmake/Findonnxruntime.cmake + DESTINATION share/fastdeploy_ppocr/cmake) +configure_package_config_file( + cmake/fastdeploy_ppocrConfig.cmake.in + ${CMAKE_CURRENT_BINARY_DIR}/fastdeploy_ppocrConfig.cmake + INSTALL_DESTINATION share/fastdeploy_ppocr) +install(FILES ${CMAKE_CURRENT_BINARY_DIR}/fastdeploy_ppocrConfig.cmake + DESTINATION share/fastdeploy_ppocr) install( DIRECTORY ${PROJECT_SOURCE_DIR}/fastdeploy DESTINATION include diff --git a/cmake/fastdeploy_ppocrConfig.cmake.in b/cmake/fastdeploy_ppocrConfig.cmake.in new file mode 100644 index 00000000000..24fe783cdfc --- /dev/null +++ b/cmake/fastdeploy_ppocrConfig.cmake.in @@ -0,0 +1,27 @@ +@PACKAGE_INIT@ + +include(CMakeFindDependencyMacro) + +# fastdeploy_ppocr's public headers need OpenCV and the exported target links +# opencv_core/opencv_imgproc, so those targets have to exist in the consumer. +find_dependency(OpenCV COMPONENTS core imgproc) + +# A static fastdeploy_ppocr carries $ in its +# interface, while a shared one does not. ONNX Runtime does not always install a +# CMake package discoverable as "onnxruntime", so look through the Find module +# shipped next to this file first. +set(fastdeploy_ppocr_STATIC @FASTDEPLOY_PPOCR_STATIC@) +if(fastdeploy_ppocr_STATIC AND NOT TARGET onnxruntime::onnxruntime) + list(APPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_LIST_DIR}/cmake") + find_dependency(onnxruntime) +endif() + +# Whether the ONNX Runtime this FastDeploy was built against ships the WebGPU +# EP. The device path itself is always compiled in and still falls back to +# Device::CPU at runtime if the EP is missing, so treat this as a hint for +# choosing devices, not as a guarantee. +set(fastdeploy_ppocr_WITH_WEBGPU @WITH_WEBGPU@) + +include("${CMAKE_CURRENT_LIST_DIR}/fastdeploy_ppocrTargets.cmake") + +check_required_components(fastdeploy_ppocr) diff --git a/fastdeploy/core/config.h b/fastdeploy/core/config.h index a427789dc1b..8a9a410ed8b 100755 --- a/fastdeploy/core/config.h +++ b/fastdeploy/core/config.h @@ -53,6 +53,10 @@ #define WITH_COREML #endif +#ifndef WITH_WEBGPU +#define WITH_WEBGPU +#endif + #ifndef ENABLE_TRT_BACKEND /* #undef ENABLE_TRT_BACKEND */ #endif diff --git a/fastdeploy/core/config.h.in b/fastdeploy/core/config.h.in index bd1ae663269..c4eae254cc9 100755 --- a/fastdeploy/core/config.h.in +++ b/fastdeploy/core/config.h.in @@ -53,6 +53,10 @@ #cmakedefine WITH_COREML #endif +#ifndef WITH_WEBGPU +#cmakedefine WITH_WEBGPU +#endif + #ifndef ENABLE_TRT_BACKEND #cmakedefine ENABLE_TRT_BACKEND #endif diff --git a/fastdeploy/fastdeploy_model.cc b/fastdeploy/fastdeploy_model.cc index 6515f1830de..5f265fd981f 100644 --- a/fastdeploy/fastdeploy_model.cc +++ b/fastdeploy/fastdeploy_model.cc @@ -83,6 +83,7 @@ bool FastDeployModel::InitRuntimeWithSpecifiedBackend() { bool use_ascend = (runtime_option.device == Device::ASCEND); bool use_directml = (runtime_option.device == Device::DIRECTML); bool use_coreml = (runtime_option.device == Device::COREML); + bool use_webgpu = (runtime_option.device == Device::WEBGPU); bool use_kunlunxin = (runtime_option.device == Device::KUNLUNXIN); if (use_cuda) { @@ -141,6 +142,13 @@ bool FastDeployModel::InitRuntimeWithSpecifiedBackend() { << runtime_option.backend << " is not supported." << std::endl; return false; } + } else if (use_webgpu) { + if (!IsSupported(valid_webgpu_backends, runtime_option.backend)) { + FDERROR << "The valid webgpu backends of model " << ModelName() + << " are " << Str(valid_webgpu_backends) << ", " + << runtime_option.backend << " is not supported." << std::endl; + return false; + } } else if (use_kunlunxin) { if (!IsSupported(valid_kunlunxin_backends, runtime_option.backend)) { FDERROR << "The valid kunlunxin backends of model " << ModelName() @@ -195,6 +203,8 @@ bool FastDeployModel::InitRuntimeWithSpecifiedDevice() { return CreateDirectMLBackend(); } else if (runtime_option.device == Device::COREML) { return CreateCoreMLBackend(); + } else if (runtime_option.device == Device::WEBGPU) { + return CreateWebGPUBackend(); } else if (runtime_option.device == Device::KUNLUNXIN) { return CreateKunlunXinBackend(); } else if (runtime_option.device == Device::SOPHGOTPUD) { @@ -209,7 +219,7 @@ bool FastDeployModel::InitRuntimeWithSpecifiedDevice() { #endif } FDERROR << "Only support " - "CPU/GPU/IPU/RKNPU/HORIZONNPU/TIMVX/KunlunXin/ASCEND/DirectML/CoreML now." + "CPU/GPU/IPU/RKNPU/HORIZONNPU/TIMVX/KunlunXin/ASCEND/DirectML/CoreML/WebGPU now." << std::endl; return false; } @@ -461,6 +471,30 @@ bool FastDeployModel::CreateCoreMLBackend() { return false; } +bool FastDeployModel::CreateWebGPUBackend() { + if (valid_webgpu_backends.size() == 0) { + FDERROR << "There's no valid webgpu backends for model: " << ModelName() + << std::endl; + return false; + } + + for (size_t i = 0; i < valid_webgpu_backends.size(); ++i) { + if (!IsBackendAvailable(valid_webgpu_backends[i])) { + continue; + } + runtime_option.backend = valid_webgpu_backends[i]; + runtime_ = std::unique_ptr(new Runtime()); + if (!runtime_->Init(runtime_option)) { + return false; + } + runtime_initialized_ = true; + return true; + } + FDERROR << "Found no valid webgpu backend for model: " << ModelName() + << std::endl; + return false; +} + bool FastDeployModel::CreateIpuBackend() { if (valid_ipu_backends.size() == 0) { FDERROR << "There's no valid ipu backends for model: " << ModelName() diff --git a/fastdeploy/fastdeploy_model.h b/fastdeploy/fastdeploy_model.h index d08aaab3b21..712ec6e535a 100755 --- a/fastdeploy/fastdeploy_model.h +++ b/fastdeploy/fastdeploy_model.h @@ -51,6 +51,9 @@ class FASTDEPLOY_DECL FastDeployModel { /** Model's valid coreml backends. This member defined all the onnxruntime coreml backends have successfully tested for the model */ std::vector valid_coreml_backends = {Backend::ORT}; + /** Model's valid webgpu backends. This member defined all the onnxruntime webgpu backends have successfully tested for the model + */ + std::vector valid_webgpu_backends = {Backend::ORT}; /** Model's valid ascend backends. This member defined all the cann backends have successfully tested for the model */ std::vector valid_ascend_backends = {}; @@ -167,6 +170,7 @@ class FASTDEPLOY_DECL FastDeployModel { bool CreateASCENDBackend(); bool CreateDirectMLBackend(); bool CreateCoreMLBackend(); + bool CreateWebGPUBackend(); bool IsSupported(const std::vector& backends, Backend backend); diff --git a/fastdeploy/runtime/backends/ort/option.h b/fastdeploy/runtime/backends/ort/option.h index 865a1e5fff5..db0119afd2d 100755 --- a/fastdeploy/runtime/backends/ort/option.h +++ b/fastdeploy/runtime/backends/ort/option.h @@ -63,7 +63,7 @@ struct OrtBackendOption { * \note Complete Takeover Semantics: If this callback is provided, OrtBackend::BuildOption * invokes it and returns immediately. All standard configurations in OrtBackendOption * (e.g., intra/inter op threads, graph optimization level, execution providers like - * DirectML/CoreML/CUDA) will be completely bypassed. The caller is responsible for + * DirectML/CoreML/CUDA/WebGPU) will be completely bypassed. The caller is responsible for * fully configuring the session_options. */ bool (*configure_session_callback)(OrtSessionOptions* session_options, void* user_data) = nullptr; diff --git a/fastdeploy/runtime/backends/ort/ort_backend.cc b/fastdeploy/runtime/backends/ort/ort_backend.cc index 03c59a665eb..56dede1d248 100644 --- a/fastdeploy/runtime/backends/ort/ort_backend.cc +++ b/fastdeploy/runtime/backends/ort/ort_backend.cc @@ -34,7 +34,18 @@ #include #endif +// The linked ONNX Runtime installs only when the +// WebGPU EP is part of that build (see get_c_cxx_api_headers() in ORT's +// cmake/onnxruntime.cmake). That file is a pure marker -- it declares nothing, +// and unlike DML/CoreML there is no dedicated WebGPU factory to call: the EP +// goes through the generic SessionOptionsAppendExecutionProvider. So probe it +// instead of including it. +#if defined(WITH_WEBGPU) && __has_include() + #define ENABLE_WEBGPU +#endif + #include +#include namespace fastdeploy { @@ -190,15 +201,69 @@ bool OrtBackend::BuildOption(const OrtBackendOption& option) { return true; } #endif +#ifdef ENABLE_WEBGPU + // If use WebGPU + else if (option.device == Device::WEBGPU) { + auto all_providers = Ort::GetAvailableProviders(); + bool support_webgpu = false; + std::string providers_msg = ""; + for (size_t i = 0; i < all_providers.size(); ++i) { + providers_msg = providers_msg + all_providers[i] + ", "; + if (all_providers[i] == "WebGpuExecutionProvider") { + support_webgpu = true; + } + } + + if (!support_webgpu) { + FDWARNING << "Compiled fastdeploy with onnxruntime doesn't " + "support WebGPU, the available providers are " + << providers_msg << "will fallback to CPUExecutionProvider." + << "Please check if onnxruntime is built with WebGPU support." + << std::endl; + option_.device = Device::CPU; + } else { + try { + // OrtSessionOptionsAppendExecutionProvider turns each key into + // "ep.webgpuexecutionprovider.", and the WebGPU EP only reads the + // camelCase key "deviceId". A snake_case "device_id" is silently + // ignored by ONNX Runtime. + std::unordered_map webgpu_options; + if (option_.device_id > 0) { + webgpu_options["deviceId"] = std::to_string(option_.device_id); + } + session_options_.AppendExecutionProvider("WebGPU", webgpu_options); + } catch (const std::exception& e) { + FDERROR << "Failed to append WebGPU execution provider: " << e.what() + << std::endl; + return false; + } + } + return true; + } +#else + // The ONNX Runtime this build links does not ship the WebGPU EP (see the + // provider factory probe at the top of this file), so the device is + // unavailable. Keep the same soft-fallback contract as the runtime check + // above, but say why. + else if (option.device == Device::WEBGPU) { + FDWARNING << "FastDeploy was built without WebGPU support: the linked " + "onnxruntime has no WebGpuExecutionProvider. Fallback to " + "CPUExecutionProvider. Rebuild against an ONNX Runtime built " + "with --use_webgpu to use Device::WEBGPU." + << std::endl; + option_.device = Device::CPU; + } +#endif return true; } bool OrtBackend::Init(const RuntimeOption& option) { if (option.device != Device::CPU && option.device != Device::CUDA && - option.device != Device::DIRECTML && option.device != Device::COREML) { + option.device != Device::DIRECTML && option.device != Device::COREML && + option.device != Device::WEBGPU) { FDERROR - << "Backend::ORT only supports Device::CPU/Device::CUDA/Device::DIRECTML/Device::COREML, but now its " + << "Backend::ORT only supports Device::CPU/Device::CUDA/Device::DIRECTML/Device::COREML/Device::WEBGPU, but now its " << option.device << "." << std::endl; return false; } @@ -530,6 +595,14 @@ void OrtBackend::InitCustomOperators() { AdaptivePool2dOp* adaptive_pool2d = new AdaptivePool2dOp{"CoreMLExecutionProvider"}; custom_operators_.push_back(adaptive_pool2d); + } else if (option_.device == Device::WEBGPU) { + // Must be the EP type name registered by ONNX Runtime + // ("WebGpuExecutionProvider"), not the "WebGPU" short name accepted by + // SessionOptions::AppendExecutionProvider: custom kernels are looked up + // by Node::GetExecutionProviderType(). + AdaptivePool2dOp* adaptive_pool2d = + new AdaptivePool2dOp{"WebGpuExecutionProvider"}; + custom_operators_.push_back(adaptive_pool2d); } else { AdaptivePool2dOp* adaptive_pool2d = new AdaptivePool2dOp{"CPUExecutionProvider"}; diff --git a/fastdeploy/runtime/backends/ort/utils.cc b/fastdeploy/runtime/backends/ort/utils.cc index f8206708800..fb8495f4cdf 100644 --- a/fastdeploy/runtime/backends/ort/utils.cc +++ b/fastdeploy/runtime/backends/ort/utils.cc @@ -60,8 +60,8 @@ FDDataType GetFdDtype(const ONNXTensorElementDataType& ort_dtype) { } Ort::Value CreateOrtValue(FDTensor& tensor) { - FDASSERT(tensor.device == Device::CUDA || tensor.device == Device::DIRECTML || tensor.device == Device::COREML || tensor.device == Device::CPU, - "Only support tensor which device is Cuda or DirectML or CPU for OrtBackend."); + FDASSERT(tensor.device == Device::CUDA || tensor.device == Device::DIRECTML || tensor.device == Device::COREML || tensor.device == Device::WEBGPU || tensor.device == Device::CPU, + "Only support tensor which device is Cuda or DirectML or CoreML or WebGPU or CPU for OrtBackend."); if (tensor.device == Device::CUDA) { Ort::MemoryInfo memory_info("Cuda", OrtDeviceAllocator, 0, OrtMemTypeDefault); @@ -78,7 +78,7 @@ Ort::Value CreateOrtValue(FDTensor& tensor) { tensor.shape.size(), GetOrtDtype(tensor.dtype)); return ort_value; } - else { // not support coreml now + else { // CoreML/WebGPU/CPU tensors live in host memory Ort::MemoryInfo memory_info("Cpu", OrtDeviceAllocator, 0, OrtMemTypeDefault); auto ort_value = Ort::Value::CreateTensor( memory_info, tensor.Data(), tensor.Nbytes(), tensor.shape.data(), diff --git a/fastdeploy/runtime/enum_variables.cc b/fastdeploy/runtime/enum_variables.cc index 7aed4313bfc..57875fe09bf 100644 --- a/fastdeploy/runtime/enum_variables.cc +++ b/fastdeploy/runtime/enum_variables.cc @@ -75,6 +75,9 @@ std::ostream& operator<<(std::ostream& out, const Device& d) { case Device::COREML: out << "Device::COREML"; break; + case Device::WEBGPU: + out << "Device::WEBGPU"; + break; default: out << "Device::UNKOWN"; } diff --git a/fastdeploy/runtime/enum_variables.h b/fastdeploy/runtime/enum_variables.h index bd9b0f77f01..7f1647e3065 100644 --- a/fastdeploy/runtime/enum_variables.h +++ b/fastdeploy/runtime/enum_variables.h @@ -30,7 +30,7 @@ namespace fastdeploy { enum Backend { UNKNOWN, ///< Unknown inference backend ORT, //< ONNX Runtime, support Paddle/ONNX format model, - //< CPU/ Nvidia GPU DirectML/CoreML + //< CPU/ Nvidia GPU DirectML/CoreML/WebGPU TRT, ///< TensorRT, support Paddle/ONNX format model, Nvidia GPU only PDINFER, ///< Paddle Inference, support Paddle format model, CPU / Nvidia GPU POROS, ///< Poros, support TorchScript format model, CPU / Nvidia GPU @@ -65,6 +65,7 @@ enum FASTDEPLOY_DECL Device { DIRECTML, COREML, SUNRISENPU, + WEBGPU, }; /*! Deep learning model format */ @@ -107,7 +108,8 @@ static std::map> {Device::ASCEND, {Backend::LITE}}, {Device::SOPHGOTPUD, {Backend::SOPHGOTPU}}, {Device::DIRECTML, {Backend::ORT}}, - {Device::COREML, {Backend::ORT}} + {Device::COREML, {Backend::ORT}}, + {Device::WEBGPU, {Backend::ORT}} }; inline bool Supported(ModelFormat format, Backend backend) { diff --git a/fastdeploy/runtime/runtime_option.cc b/fastdeploy/runtime/runtime_option.cc index 66c0cd6fff8..b9e12b59366 100644 --- a/fastdeploy/runtime/runtime_option.cc +++ b/fastdeploy/runtime/runtime_option.cc @@ -151,6 +151,11 @@ void RuntimeOption::UseCoreML(uint32_t coreml_flag) { device_id = coreml_flag; } +void RuntimeOption::UseWebGPU(int device_id) { + device = Device::WEBGPU; + this->device_id = device_id; +} + void RuntimeOption::UseSophgo() { device = Device::SOPHGOTPUD; UseSophgoBackend(); diff --git a/fastdeploy/runtime/runtime_option.h b/fastdeploy/runtime/runtime_option.h index e371d0bb5f7..a237206dcbd 100755 --- a/fastdeploy/runtime/runtime_option.h +++ b/fastdeploy/runtime/runtime_option.h @@ -87,6 +87,9 @@ struct FASTDEPLOY_DECL RuntimeOption { /// Use onnxruntime CoreML to inference void UseCoreML(uint32_t coreml_flag = 0); + /// Use onnxruntime WebGPU to inference + void UseWebGPU(int device_id = 0); + /// Use Sophgo to inference void UseSophgo(); /// \brief Turn on KunlunXin XPU. diff --git a/fastdeploy/vision/ocr/ppocr/structurev2_ser_vi_layoutxlm.cc b/fastdeploy/vision/ocr/ppocr/structurev2_ser_vi_layoutxlm.cc index e9300d8577c..b5a031a2b99 100644 --- a/fastdeploy/vision/ocr/ppocr/structurev2_ser_vi_layoutxlm.cc +++ b/fastdeploy/vision/ocr/ppocr/structurev2_ser_vi_layoutxlm.cc @@ -34,6 +34,7 @@ StructureV2SERViLayoutXLMModel::StructureV2SERViLayoutXLMModel( valid_ipu_backends = {Backend::PDINFER}; valid_directml_backends = {Backend::ORT}; valid_coreml_backends = {Backend::ORT}; + valid_webgpu_backends = {Backend::ORT}; } else if (model_format == ModelFormat::SOPHGO) { valid_sophgonpu_backends = {Backend::SOPHGOTPU}; } else { @@ -42,6 +43,7 @@ StructureV2SERViLayoutXLMModel::StructureV2SERViLayoutXLMModel( valid_rknpu_backends = {Backend::RKNPU2}; valid_directml_backends = {Backend::ORT}; valid_coreml_backends = {Backend::ORT}; + valid_webgpu_backends = {Backend::ORT}; valid_horizon_backends = {Backend::HORIZONNPU}; }