From 372d4cce8b1d775e634fe27dc476b3f9f54885e8 Mon Sep 17 00:00:00 2001 From: SherkeyXD <57581480+SherkeyXD@users.noreply.github.com> Date: Thu, 17 Sep 2026 17:18:14 +0800 Subject: [PATCH 1/3] =?UTF-8?q?feat:=20=E5=BC=80=E6=94=BE=20webgpu=20?= =?UTF-8?q?=E6=8E=A8=E7=90=86=E9=85=8D=E7=BD=AE?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- CMakeLists.txt | 24 ++++++++- cmake/fastdeploy_ppocrConfig.cmake.in | 13 +++++ fastdeploy/core/config.h | 4 ++ fastdeploy/core/config.h.in | 4 ++ fastdeploy/fastdeploy_model.cc | 34 +++++++++++++ fastdeploy/fastdeploy_model.h | 4 ++ .../runtime/backends/ort/ort_backend.cc | 51 ++++++++++++++++++- fastdeploy/runtime/backends/ort/utils.cc | 4 +- fastdeploy/runtime/enum_variables.cc | 3 ++ fastdeploy/runtime/enum_variables.h | 4 +- fastdeploy/runtime/runtime_option.cc | 5 ++ fastdeploy/runtime/runtime_option.h | 3 ++ .../ocr/ppocr/structurev2_ser_vi_layoutxlm.cc | 2 + 13 files changed, 148 insertions(+), 7 deletions(-) create mode 100644 cmake/fastdeploy_ppocrConfig.cmake.in diff --git a/CMakeLists.txt b/CMakeLists.txt index f662b8c22e5..a3401799935 100755 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -117,9 +117,20 @@ set_target_properties( target_include_directories(fastdeploy_ppocr INTERFACE $ # for build - $ # for install + # Headers are installed to /include (see the install() below). + # CMAKE_INSTALL_INCLUDE is not defined in this project, so spell the + # relative path out: an empty INSTALL_INTERFACE leaves the imported target + # without any include directory for consumers. + $ # for install ) +# WITH_WEBGPU is a capability of this source tree, not a build switch: the +# WebGPU device is always compiled in and gated at runtime by +# Ort::GetAvailableProviders() (the macro is declared in core/config.h +# alongside WITH_DIRECTML/WITH_COREML). Exporting it as a PUBLIC definition +# lets consumers check it as well. +target_compile_definitions(fastdeploy_ppocr PUBLIC WITH_WEBGPU) + target_link_libraries(fastdeploy_ppocr PUBLIC ${OpenCV_LIBS} PRIVATE onnxruntime::onnxruntime) if(WITH_CUDA) target_link_libraries(fastdeploy_ppocr PRIVATE ${CUDA_LIB}) @@ -129,7 +140,16 @@ if(ANDROID) endif() install(TARGETS fastdeploy_ppocr EXPORT fastdeploy_ppocrConfig) -install(EXPORT fastdeploy_ppocrConfig DESTINATION share/fastdeploy_ppocr) +include(CMakePackageConfigHelpers) +install(EXPORT fastdeploy_ppocrConfig + DESTINATION share/fastdeploy_ppocr + FILE fastdeploy_ppocrTargets.cmake) +configure_package_config_file( + cmake/fastdeploy_ppocrConfig.cmake.in + ${CMAKE_CURRENT_BINARY_DIR}/fastdeploy_ppocrConfig.cmake + INSTALL_DESTINATION share/fastdeploy_ppocr) +install(FILES ${CMAKE_CURRENT_BINARY_DIR}/fastdeploy_ppocrConfig.cmake + DESTINATION share/fastdeploy_ppocr) install( DIRECTORY ${PROJECT_SOURCE_DIR}/fastdeploy DESTINATION include diff --git a/cmake/fastdeploy_ppocrConfig.cmake.in b/cmake/fastdeploy_ppocrConfig.cmake.in new file mode 100644 index 00000000000..d3effc087db --- /dev/null +++ b/cmake/fastdeploy_ppocrConfig.cmake.in @@ -0,0 +1,13 @@ +@PACKAGE_INIT@ + +# Capabilities of the FastDeploy build that produced this package, so that +# consumers can tell which devices the library was built with instead of +# probing at runtime. WITH_WEBGPU mirrors fastdeploy/core/config.h: the WebGPU +# device is always compiled in and gated at runtime by +# Ort::GetAvailableProviders(), so it is a capability flag rather than a build +# switch (same as WITH_DIRECTML/WITH_COREML). +set(WITH_WEBGPU ON) + +include("${CMAKE_CURRENT_LIST_DIR}/fastdeploy_ppocrTargets.cmake") + +check_required_components(fastdeploy_ppocr) diff --git a/fastdeploy/core/config.h b/fastdeploy/core/config.h index a427789dc1b..8a9a410ed8b 100755 --- a/fastdeploy/core/config.h +++ b/fastdeploy/core/config.h @@ -53,6 +53,10 @@ #define WITH_COREML #endif +#ifndef WITH_WEBGPU +#define WITH_WEBGPU +#endif + #ifndef ENABLE_TRT_BACKEND /* #undef ENABLE_TRT_BACKEND */ #endif diff --git a/fastdeploy/core/config.h.in b/fastdeploy/core/config.h.in index bd1ae663269..c4eae254cc9 100755 --- a/fastdeploy/core/config.h.in +++ b/fastdeploy/core/config.h.in @@ -53,6 +53,10 @@ #cmakedefine WITH_COREML #endif +#ifndef WITH_WEBGPU +#cmakedefine WITH_WEBGPU +#endif + #ifndef ENABLE_TRT_BACKEND #cmakedefine ENABLE_TRT_BACKEND #endif diff --git a/fastdeploy/fastdeploy_model.cc b/fastdeploy/fastdeploy_model.cc index 6515f1830de..a7c15214ae6 100644 --- a/fastdeploy/fastdeploy_model.cc +++ b/fastdeploy/fastdeploy_model.cc @@ -83,6 +83,7 @@ bool FastDeployModel::InitRuntimeWithSpecifiedBackend() { bool use_ascend = (runtime_option.device == Device::ASCEND); bool use_directml = (runtime_option.device == Device::DIRECTML); bool use_coreml = (runtime_option.device == Device::COREML); + bool use_webgpu = (runtime_option.device == Device::WEBGPU); bool use_kunlunxin = (runtime_option.device == Device::KUNLUNXIN); if (use_cuda) { @@ -141,6 +142,13 @@ bool FastDeployModel::InitRuntimeWithSpecifiedBackend() { << runtime_option.backend << " is not supported." << std::endl; return false; } + } else if (use_webgpu) { + if (!IsSupported(valid_webgpu_backends, runtime_option.backend)) { + FDERROR << "The valid webgpu backends of model " << ModelName() + << " are " << Str(valid_webgpu_backends) << ", " + << runtime_option.backend << " is not supported." << std::endl; + return false; + } } else if (use_kunlunxin) { if (!IsSupported(valid_kunlunxin_backends, runtime_option.backend)) { FDERROR << "The valid kunlunxin backends of model " << ModelName() @@ -195,6 +203,8 @@ bool FastDeployModel::InitRuntimeWithSpecifiedDevice() { return CreateDirectMLBackend(); } else if (runtime_option.device == Device::COREML) { return CreateCoreMLBackend(); + } else if (runtime_option.device == Device::WEBGPU) { + return CreateWebGPUBackend(); } else if (runtime_option.device == Device::KUNLUNXIN) { return CreateKunlunXinBackend(); } else if (runtime_option.device == Device::SOPHGOTPUD) { @@ -461,6 +471,30 @@ bool FastDeployModel::CreateCoreMLBackend() { return false; } +bool FastDeployModel::CreateWebGPUBackend() { + if (valid_webgpu_backends.size() == 0) { + FDERROR << "There's no valid webgpu backends for model: " << ModelName() + << std::endl; + return false; + } + + for (size_t i = 0; i < valid_webgpu_backends.size(); ++i) { + if (!IsBackendAvailable(valid_webgpu_backends[i])) { + continue; + } + runtime_option.backend = valid_webgpu_backends[i]; + runtime_ = std::unique_ptr(new Runtime()); + if (!runtime_->Init(runtime_option)) { + return false; + } + runtime_initialized_ = true; + return true; + } + FDERROR << "Found no valid webgpu backend for model: " << ModelName() + << std::endl; + return false; +} + bool FastDeployModel::CreateIpuBackend() { if (valid_ipu_backends.size() == 0) { FDERROR << "There's no valid ipu backends for model: " << ModelName() diff --git a/fastdeploy/fastdeploy_model.h b/fastdeploy/fastdeploy_model.h index d08aaab3b21..712ec6e535a 100755 --- a/fastdeploy/fastdeploy_model.h +++ b/fastdeploy/fastdeploy_model.h @@ -51,6 +51,9 @@ class FASTDEPLOY_DECL FastDeployModel { /** Model's valid coreml backends. This member defined all the onnxruntime coreml backends have successfully tested for the model */ std::vector valid_coreml_backends = {Backend::ORT}; + /** Model's valid webgpu backends. This member defined all the onnxruntime webgpu backends have successfully tested for the model + */ + std::vector valid_webgpu_backends = {Backend::ORT}; /** Model's valid ascend backends. This member defined all the cann backends have successfully tested for the model */ std::vector valid_ascend_backends = {}; @@ -167,6 +170,7 @@ class FASTDEPLOY_DECL FastDeployModel { bool CreateASCENDBackend(); bool CreateDirectMLBackend(); bool CreateCoreMLBackend(); + bool CreateWebGPUBackend(); bool IsSupported(const std::vector& backends, Backend backend); diff --git a/fastdeploy/runtime/backends/ort/ort_backend.cc b/fastdeploy/runtime/backends/ort/ort_backend.cc index 03c59a665eb..90189d5bd5a 100644 --- a/fastdeploy/runtime/backends/ort/ort_backend.cc +++ b/fastdeploy/runtime/backends/ort/ort_backend.cc @@ -190,15 +190,54 @@ bool OrtBackend::BuildOption(const OrtBackendOption& option) { return true; } #endif + // If use WebGPU + else if (option.device == Device::WEBGPU) { + auto all_providers = Ort::GetAvailableProviders(); + bool support_webgpu = false; + std::string providers_msg = ""; + for (size_t i = 0; i < all_providers.size(); ++i) { + providers_msg = providers_msg + all_providers[i] + ", "; + if (all_providers[i] == "WebGpuExecutionProvider") { + support_webgpu = true; + } + } + + if (!support_webgpu) { + FDWARNING << "Compiled fastdeploy with onnxruntime doesn't " + "support WebGPU, the available providers are " + << providers_msg << "will fallback to CPUExecutionProvider." + << "Please check if onnxruntime is built with WebGPU support." + << std::endl; + option_.device = Device::CPU; + } else { + try { + // OrtSessionOptionsAppendExecutionProvider turns each key into + // "ep.webgpuexecutionprovider.", and the WebGPU EP only reads the + // camelCase key "deviceId". A snake_case "device_id" is silently + // ignored by ONNX Runtime. + std::unordered_map webgpu_options; + if (option_.device_id > 0) { + webgpu_options["deviceId"] = std::to_string(option_.device_id); + } + session_options_.AppendExecutionProvider("WebGPU", webgpu_options); + } catch (const std::exception& e) { + FDERROR << "Failed to append WebGPU execution provider: " << e.what() + << std::endl; + return false; + } + } + return true; + } return true; } bool OrtBackend::Init(const RuntimeOption& option) { if (option.device != Device::CPU && option.device != Device::CUDA && - option.device != Device::DIRECTML && option.device != Device::COREML) { + option.device != Device::DIRECTML && option.device != Device::COREML && + option.device != Device::WEBGPU) { FDERROR - << "Backend::ORT only supports Device::CPU/Device::CUDA/Device::DIRECTML/Device::COREML, but now its " + << "Backend::ORT only supports Device::CPU/Device::CUDA/Device::DIRECTML/Device::COREML/Device::WEBGPU, but now its " << option.device << "." << std::endl; return false; } @@ -530,6 +569,14 @@ void OrtBackend::InitCustomOperators() { AdaptivePool2dOp* adaptive_pool2d = new AdaptivePool2dOp{"CoreMLExecutionProvider"}; custom_operators_.push_back(adaptive_pool2d); + } else if (option_.device == Device::WEBGPU) { + // Must be the EP type name registered by ONNX Runtime + // ("WebGpuExecutionProvider"), not the "WebGPU" short name accepted by + // SessionOptions::AppendExecutionProvider: custom kernels are looked up + // by Node::GetExecutionProviderType(). + AdaptivePool2dOp* adaptive_pool2d = + new AdaptivePool2dOp{"WebGpuExecutionProvider"}; + custom_operators_.push_back(adaptive_pool2d); } else { AdaptivePool2dOp* adaptive_pool2d = new AdaptivePool2dOp{"CPUExecutionProvider"}; diff --git a/fastdeploy/runtime/backends/ort/utils.cc b/fastdeploy/runtime/backends/ort/utils.cc index f8206708800..4a4e595a45f 100644 --- a/fastdeploy/runtime/backends/ort/utils.cc +++ b/fastdeploy/runtime/backends/ort/utils.cc @@ -60,8 +60,8 @@ FDDataType GetFdDtype(const ONNXTensorElementDataType& ort_dtype) { } Ort::Value CreateOrtValue(FDTensor& tensor) { - FDASSERT(tensor.device == Device::CUDA || tensor.device == Device::DIRECTML || tensor.device == Device::COREML || tensor.device == Device::CPU, - "Only support tensor which device is Cuda or DirectML or CPU for OrtBackend."); + FDASSERT(tensor.device == Device::CUDA || tensor.device == Device::DIRECTML || tensor.device == Device::COREML || tensor.device == Device::WEBGPU || tensor.device == Device::CPU, + "Only support tensor which device is Cuda or DirectML or CoreML or WebGPU or CPU for OrtBackend."); if (tensor.device == Device::CUDA) { Ort::MemoryInfo memory_info("Cuda", OrtDeviceAllocator, 0, OrtMemTypeDefault); diff --git a/fastdeploy/runtime/enum_variables.cc b/fastdeploy/runtime/enum_variables.cc index 7aed4313bfc..57875fe09bf 100644 --- a/fastdeploy/runtime/enum_variables.cc +++ b/fastdeploy/runtime/enum_variables.cc @@ -75,6 +75,9 @@ std::ostream& operator<<(std::ostream& out, const Device& d) { case Device::COREML: out << "Device::COREML"; break; + case Device::WEBGPU: + out << "Device::WEBGPU"; + break; default: out << "Device::UNKOWN"; } diff --git a/fastdeploy/runtime/enum_variables.h b/fastdeploy/runtime/enum_variables.h index bd9b0f77f01..9af93554511 100644 --- a/fastdeploy/runtime/enum_variables.h +++ b/fastdeploy/runtime/enum_variables.h @@ -65,6 +65,7 @@ enum FASTDEPLOY_DECL Device { DIRECTML, COREML, SUNRISENPU, + WEBGPU, }; /*! Deep learning model format */ @@ -107,7 +108,8 @@ static std::map> {Device::ASCEND, {Backend::LITE}}, {Device::SOPHGOTPUD, {Backend::SOPHGOTPU}}, {Device::DIRECTML, {Backend::ORT}}, - {Device::COREML, {Backend::ORT}} + {Device::COREML, {Backend::ORT}}, + {Device::WEBGPU, {Backend::ORT}} }; inline bool Supported(ModelFormat format, Backend backend) { diff --git a/fastdeploy/runtime/runtime_option.cc b/fastdeploy/runtime/runtime_option.cc index 66c0cd6fff8..b9e12b59366 100644 --- a/fastdeploy/runtime/runtime_option.cc +++ b/fastdeploy/runtime/runtime_option.cc @@ -151,6 +151,11 @@ void RuntimeOption::UseCoreML(uint32_t coreml_flag) { device_id = coreml_flag; } +void RuntimeOption::UseWebGPU(int device_id) { + device = Device::WEBGPU; + this->device_id = device_id; +} + void RuntimeOption::UseSophgo() { device = Device::SOPHGOTPUD; UseSophgoBackend(); diff --git a/fastdeploy/runtime/runtime_option.h b/fastdeploy/runtime/runtime_option.h index e371d0bb5f7..a237206dcbd 100755 --- a/fastdeploy/runtime/runtime_option.h +++ b/fastdeploy/runtime/runtime_option.h @@ -87,6 +87,9 @@ struct FASTDEPLOY_DECL RuntimeOption { /// Use onnxruntime CoreML to inference void UseCoreML(uint32_t coreml_flag = 0); + /// Use onnxruntime WebGPU to inference + void UseWebGPU(int device_id = 0); + /// Use Sophgo to inference void UseSophgo(); /// \brief Turn on KunlunXin XPU. diff --git a/fastdeploy/vision/ocr/ppocr/structurev2_ser_vi_layoutxlm.cc b/fastdeploy/vision/ocr/ppocr/structurev2_ser_vi_layoutxlm.cc index e9300d8577c..b5a031a2b99 100644 --- a/fastdeploy/vision/ocr/ppocr/structurev2_ser_vi_layoutxlm.cc +++ b/fastdeploy/vision/ocr/ppocr/structurev2_ser_vi_layoutxlm.cc @@ -34,6 +34,7 @@ StructureV2SERViLayoutXLMModel::StructureV2SERViLayoutXLMModel( valid_ipu_backends = {Backend::PDINFER}; valid_directml_backends = {Backend::ORT}; valid_coreml_backends = {Backend::ORT}; + valid_webgpu_backends = {Backend::ORT}; } else if (model_format == ModelFormat::SOPHGO) { valid_sophgonpu_backends = {Backend::SOPHGOTPU}; } else { @@ -42,6 +43,7 @@ StructureV2SERViLayoutXLMModel::StructureV2SERViLayoutXLMModel( valid_rknpu_backends = {Backend::RKNPU2}; valid_directml_backends = {Backend::ORT}; valid_coreml_backends = {Backend::ORT}; + valid_webgpu_backends = {Backend::ORT}; valid_horizon_backends = {Backend::HORIZONNPU}; } From 55fdc946768041f24e1a5250b63b95f857324f27 Mon Sep 17 00:00:00 2001 From: SherkeyXD <57581480+SherkeyXD@users.noreply.github.com> Date: Fri, 18 Sep 2026 15:35:19 +0800 Subject: [PATCH 2/3] =?UTF-8?q?feat:=20webgpu=20=E6=94=AF=E6=8C=81?= =?UTF-8?q?=E6=8C=89=20onnxruntime=20=E8=83=BD=E5=8A=9B=E6=8E=A2=E6=B5=8B?= =?UTF-8?q?=E5=B9=B6=E6=8C=89=E9=9C=80=E7=BC=96=E8=AF=91?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit onnxruntime 只在该 EP 参与构建时才把 平铺安装出来 (cmake/onnxruntime.cmake 按 ONNXRUNTIME_PROVIDER_NAMES 逐个 provider glob), 且这个头文件只有注释、没有声明,WebGPU 走的是 generic SessionOptionsAppendExecutionProvider,因此用它做 __has_include 探测即可: - WITH_WEBGPU && __has_include() => ENABLE_WEBGPU - BuildOption 的 WebGPU 分支加 #ifdef ENABLE_WEBGPU 守卫;未编入时告警并回退 CPUExecutionProvider,沿用 DML/CoreML 的软回退契约。此时 option_.device 已被 置为 CPU,后续 InitCustomOperators 会正确注册 CPU 版自定义算子 - 补齐 WebGPU 在错误信息/注释中的出现,并为 std::unordered_map 补显式 include --- fastdeploy/fastdeploy_model.cc | 2 +- fastdeploy/runtime/backends/ort/option.h | 2 +- .../runtime/backends/ort/ort_backend.cc | 26 +++++++++++++++++++ fastdeploy/runtime/backends/ort/utils.cc | 2 +- fastdeploy/runtime/enum_variables.h | 2 +- 5 files changed, 30 insertions(+), 4 deletions(-) diff --git a/fastdeploy/fastdeploy_model.cc b/fastdeploy/fastdeploy_model.cc index a7c15214ae6..5f265fd981f 100644 --- a/fastdeploy/fastdeploy_model.cc +++ b/fastdeploy/fastdeploy_model.cc @@ -219,7 +219,7 @@ bool FastDeployModel::InitRuntimeWithSpecifiedDevice() { #endif } FDERROR << "Only support " - "CPU/GPU/IPU/RKNPU/HORIZONNPU/TIMVX/KunlunXin/ASCEND/DirectML/CoreML now." + "CPU/GPU/IPU/RKNPU/HORIZONNPU/TIMVX/KunlunXin/ASCEND/DirectML/CoreML/WebGPU now." << std::endl; return false; } diff --git a/fastdeploy/runtime/backends/ort/option.h b/fastdeploy/runtime/backends/ort/option.h index 865a1e5fff5..db0119afd2d 100755 --- a/fastdeploy/runtime/backends/ort/option.h +++ b/fastdeploy/runtime/backends/ort/option.h @@ -63,7 +63,7 @@ struct OrtBackendOption { * \note Complete Takeover Semantics: If this callback is provided, OrtBackend::BuildOption * invokes it and returns immediately. All standard configurations in OrtBackendOption * (e.g., intra/inter op threads, graph optimization level, execution providers like - * DirectML/CoreML/CUDA) will be completely bypassed. The caller is responsible for + * DirectML/CoreML/CUDA/WebGPU) will be completely bypassed. The caller is responsible for * fully configuring the session_options. */ bool (*configure_session_callback)(OrtSessionOptions* session_options, void* user_data) = nullptr; diff --git a/fastdeploy/runtime/backends/ort/ort_backend.cc b/fastdeploy/runtime/backends/ort/ort_backend.cc index 90189d5bd5a..56dede1d248 100644 --- a/fastdeploy/runtime/backends/ort/ort_backend.cc +++ b/fastdeploy/runtime/backends/ort/ort_backend.cc @@ -34,7 +34,18 @@ #include #endif +// The linked ONNX Runtime installs only when the +// WebGPU EP is part of that build (see get_c_cxx_api_headers() in ORT's +// cmake/onnxruntime.cmake). That file is a pure marker -- it declares nothing, +// and unlike DML/CoreML there is no dedicated WebGPU factory to call: the EP +// goes through the generic SessionOptionsAppendExecutionProvider. So probe it +// instead of including it. +#if defined(WITH_WEBGPU) && __has_include() + #define ENABLE_WEBGPU +#endif + #include +#include namespace fastdeploy { @@ -190,6 +201,7 @@ bool OrtBackend::BuildOption(const OrtBackendOption& option) { return true; } #endif +#ifdef ENABLE_WEBGPU // If use WebGPU else if (option.device == Device::WEBGPU) { auto all_providers = Ort::GetAvailableProviders(); @@ -228,6 +240,20 @@ bool OrtBackend::BuildOption(const OrtBackendOption& option) { } return true; } +#else + // The ONNX Runtime this build links does not ship the WebGPU EP (see the + // provider factory probe at the top of this file), so the device is + // unavailable. Keep the same soft-fallback contract as the runtime check + // above, but say why. + else if (option.device == Device::WEBGPU) { + FDWARNING << "FastDeploy was built without WebGPU support: the linked " + "onnxruntime has no WebGpuExecutionProvider. Fallback to " + "CPUExecutionProvider. Rebuild against an ONNX Runtime built " + "with --use_webgpu to use Device::WEBGPU." + << std::endl; + option_.device = Device::CPU; + } +#endif return true; } diff --git a/fastdeploy/runtime/backends/ort/utils.cc b/fastdeploy/runtime/backends/ort/utils.cc index 4a4e595a45f..fb8495f4cdf 100644 --- a/fastdeploy/runtime/backends/ort/utils.cc +++ b/fastdeploy/runtime/backends/ort/utils.cc @@ -78,7 +78,7 @@ Ort::Value CreateOrtValue(FDTensor& tensor) { tensor.shape.size(), GetOrtDtype(tensor.dtype)); return ort_value; } - else { // not support coreml now + else { // CoreML/WebGPU/CPU tensors live in host memory Ort::MemoryInfo memory_info("Cpu", OrtDeviceAllocator, 0, OrtMemTypeDefault); auto ort_value = Ort::Value::CreateTensor( memory_info, tensor.Data(), tensor.Nbytes(), tensor.shape.data(), diff --git a/fastdeploy/runtime/enum_variables.h b/fastdeploy/runtime/enum_variables.h index 9af93554511..7f1647e3065 100644 --- a/fastdeploy/runtime/enum_variables.h +++ b/fastdeploy/runtime/enum_variables.h @@ -30,7 +30,7 @@ namespace fastdeploy { enum Backend { UNKNOWN, ///< Unknown inference backend ORT, //< ONNX Runtime, support Paddle/ONNX format model, - //< CPU/ Nvidia GPU DirectML/CoreML + //< CPU/ Nvidia GPU DirectML/CoreML/WebGPU TRT, ///< TensorRT, support Paddle/ONNX format model, Nvidia GPU only PDINFER, ///< Paddle Inference, support Paddle format model, CPU / Nvidia GPU POROS, ///< Poros, support TorchScript format model, CPU / Nvidia GPU From 36c31cc8674251614bf135e5eead0d13f83203ba Mon Sep 17 00:00:00 2001 From: SherkeyXD <57581480+SherkeyXD@users.noreply.github.com> Date: Fri, 18 Sep 2026 15:35:19 +0800 Subject: [PATCH 3/3] =?UTF-8?q?build:=20=E5=AF=BC=E5=87=BA=20fastdeploy=5F?= =?UTF-8?q?ppocr=20=E7=9A=84=E5=8C=85=E9=85=8D=E7=BD=AE=E4=B8=8E=E4=BE=9D?= =?UTF-8?q?=E8=B5=96?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - install(EXPORT ... FILE fastdeploy_ppocrTargets.cmake) 搭配 configure_package_config_file 生成可编辑的 fastdeploy_ppocrConfig.cmake - find_dependency(OpenCV COMPONENTS core imgproc):导出的 target 链接 opencv_core/opencv_imgproc,下游若不先 find_package(OpenCV) 会在链接期报 ld: library 'opencv_core' not found - 静态包的 interface 带 $,故仅静态时 find_dependency(onnxruntime),并随包安装 cmake/Findonnxruntime.cmake,避免 依赖 onnxruntime 的 CMake 包恰好能被该名字找到 - $:CMAKE_INSTALL_INCLUDE 在本项目从未定义,原先 导出的 target 没有任何 include 目录 - 导出 fastdeploy_ppocr_WITH_WEBGPU 能力标记,取值与编译期探测同源 --- CMakeLists.txt | 32 +++++++++++++++++++++------ cmake/fastdeploy_ppocrConfig.cmake.in | 28 +++++++++++++++++------ 2 files changed, 46 insertions(+), 14 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index a3401799935..17bf24485d6 100755 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -8,6 +8,20 @@ find_package(onnxruntime REQUIRED) find_package(Eigen3 REQUIRED) message("OpenCV_LIBS: ${OpenCV_LIBS} ${OPENCV_core_FOUND} ${OPENCV_WORLD_FOUND}") +# ONNX Runtime flattens the headers of every enabled execution provider into +# , so exists exactly when +# that ONNX Runtime was built with the WebGPU EP (the same signal the Maa +# side uses for DirectML/CoreML). FastDeploy always compiles the WebGPU device +# path and falls back to Device::CPU at runtime when the EP is unavailable, so +# this reports what the linked ONNX Runtime can actually do -- pick the ONNX +# Runtime accordingly if you need the device. +if(EXISTS "${onnxruntime_INCLUDE_DIR}/webgpu_provider_factory.h") + set(WITH_WEBGPU ON) +else() + set(WITH_WEBGPU OFF) +endif() +message(STATUS "WITH_WEBGPU: ${WITH_WEBGPU} (onnxruntime_INCLUDE_DIR=${onnxruntime_INCLUDE_DIR})") + option(WITH_CUDA "Whether WITH_CUDA=ON, will enable onnxruntime-gpu/paddle-inference-gpu" OFF) option(PRINT_INFO "Print more debug info while running" OFF) @@ -124,13 +138,6 @@ target_include_directories(fastdeploy_ppocr INTERFACE $ # for install ) -# WITH_WEBGPU is a capability of this source tree, not a build switch: the -# WebGPU device is always compiled in and gated at runtime by -# Ort::GetAvailableProviders() (the macro is declared in core/config.h -# alongside WITH_DIRECTML/WITH_COREML). Exporting it as a PUBLIC definition -# lets consumers check it as well. -target_compile_definitions(fastdeploy_ppocr PUBLIC WITH_WEBGPU) - target_link_libraries(fastdeploy_ppocr PUBLIC ${OpenCV_LIBS} PRIVATE onnxruntime::onnxruntime) if(WITH_CUDA) target_link_libraries(fastdeploy_ppocr PRIVATE ${CUDA_LIB}) @@ -144,6 +151,17 @@ include(CMakePackageConfigHelpers) install(EXPORT fastdeploy_ppocrConfig DESTINATION share/fastdeploy_ppocr FILE fastdeploy_ppocrTargets.cmake) +# A static build leaks $ into the exported +# interface, so the package config has to help consumers materialize that +# target; ship the Find module it needs. +get_target_property(FASTDEPLOY_PPOCR_LIBRARY_TYPE fastdeploy_ppocr TYPE) +if(FASTDEPLOY_PPOCR_LIBRARY_TYPE STREQUAL "STATIC_LIBRARY") + set(FASTDEPLOY_PPOCR_STATIC ON) +else() + set(FASTDEPLOY_PPOCR_STATIC OFF) +endif() +install(FILES cmake/Findonnxruntime.cmake + DESTINATION share/fastdeploy_ppocr/cmake) configure_package_config_file( cmake/fastdeploy_ppocrConfig.cmake.in ${CMAKE_CURRENT_BINARY_DIR}/fastdeploy_ppocrConfig.cmake diff --git a/cmake/fastdeploy_ppocrConfig.cmake.in b/cmake/fastdeploy_ppocrConfig.cmake.in index d3effc087db..24fe783cdfc 100644 --- a/cmake/fastdeploy_ppocrConfig.cmake.in +++ b/cmake/fastdeploy_ppocrConfig.cmake.in @@ -1,12 +1,26 @@ @PACKAGE_INIT@ -# Capabilities of the FastDeploy build that produced this package, so that -# consumers can tell which devices the library was built with instead of -# probing at runtime. WITH_WEBGPU mirrors fastdeploy/core/config.h: the WebGPU -# device is always compiled in and gated at runtime by -# Ort::GetAvailableProviders(), so it is a capability flag rather than a build -# switch (same as WITH_DIRECTML/WITH_COREML). -set(WITH_WEBGPU ON) +include(CMakeFindDependencyMacro) + +# fastdeploy_ppocr's public headers need OpenCV and the exported target links +# opencv_core/opencv_imgproc, so those targets have to exist in the consumer. +find_dependency(OpenCV COMPONENTS core imgproc) + +# A static fastdeploy_ppocr carries $ in its +# interface, while a shared one does not. ONNX Runtime does not always install a +# CMake package discoverable as "onnxruntime", so look through the Find module +# shipped next to this file first. +set(fastdeploy_ppocr_STATIC @FASTDEPLOY_PPOCR_STATIC@) +if(fastdeploy_ppocr_STATIC AND NOT TARGET onnxruntime::onnxruntime) + list(APPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_LIST_DIR}/cmake") + find_dependency(onnxruntime) +endif() + +# Whether the ONNX Runtime this FastDeploy was built against ships the WebGPU +# EP. The device path itself is always compiled in and still falls back to +# Device::CPU at runtime if the EP is missing, so treat this as a hint for +# choosing devices, not as a guarantee. +set(fastdeploy_ppocr_WITH_WEBGPU @WITH_WEBGPU@) include("${CMAKE_CURRENT_LIST_DIR}/fastdeploy_ppocrTargets.cmake")